diff --git a/Dockerfile.postprofile b/Dockerfile.postprofile new file mode 100644 index 0000000..9d87674 --- /dev/null +++ b/Dockerfile.postprofile @@ -0,0 +1,14 @@ +FROM h3-blackwell-runtime:post-fc2-profiled-a29b896 + +WORKDIR /opt/h3-blackwell-runtime + +# Preserve the profiled binary/ABI layers while making benchmark diagnostics opt-in. +COPY src/h3_blackwell_runtime/runtime.py src/h3_blackwell_runtime/runtime.py +COPY src/h3_blackwell_runtime/sampler.py src/h3_blackwell_runtime/sampler.py +COPY tools/serve_hot_runtime.py tools/serve_hot_runtime.py +COPY tools/benchmark_hot_runtime.py tools/benchmark_hot_runtime.py +COPY tools/profile_sampling_stages.py tools/profile_sampling_stages.py +COPY tools/summarize_nsys_profile.py tools/summarize_nsys_profile.py +COPY tools/summarize_ncu_profile.py tools/summarize_ncu_profile.py +COPY tools/build_post_fc2_profile_summary.py tools/build_post_fc2_profile_summary.py +COPY tests/test_turbo.py tests/test_turbo.py diff --git a/benchmarks/gb10-post-fc2-block24-profile-20260826.json b/benchmarks/gb10-post-fc2-block24-profile-20260826.json new file mode 100644 index 0000000..a0618bb --- /dev/null +++ b/benchmarks/gb10-post-fc2-block24-profile-20260826.json @@ -0,0 +1,1302 @@ +{ + "prompt": "A playful orange tabby cat starts in an ordinary cozy living room in a normal house, afternoon light, sofa and rug. The cat crouches, jumps, and does one clean athletic backflip in slow motion. As the backflip completes there is a sharp cinematic cut: the cat lands perfectly on a glowing neon disco dance floor wearing oversized black sunglasses. Mirror ball reflections, colorful lights, joyful party energy, stylish and funny, clear before-and-after transformation.", + "model_path": "/models/minimax_h3_fl2va_pruned_nvfp4.safetensors", + "width": 1344, + "height": 768, + "frames": 124, + "steps": 12, + "sampler_step": 1, + "seed": 440420, + "text_tokens": 100, + "block_index": 24, + "attention": "sage2", + "warmup": 20, + "iterations": 50, + "hidden_shape": [ + 37810, + 5376 + ], + "output_shape": [ + 37810, + 5376 + ], + "segments": [ + [ + 0, + 100, + 1 + ], + [ + 100, + 514, + 2 + ], + [ + 514, + 37810, + 0 + ] + ], + "timings": { + "norm1": { + "count": 50, + "mean_s": 0.0035359218600569875, + "p50_s": 0.0035142429996994906, + "p90_s": 0.003590562999852409, + "p95_s": 0.0036120837999078503, + "p99_s": 0.003746741560207738, + "min_s": 0.0034944029994221637, + "max_s": 0.003793122999923071 + }, + "modulate_msa": { + "count": 50, + "mean_s": 0.010587173820022144, + "p50_s": 0.01043039999967732, + "p90_s": 0.010989329001313308, + "p95_s": 0.010999922849896393, + "p99_s": 0.011023483050266805, + "min_s": 0.010271736000504461, + "max_s": 0.011034584000299219 + }, + "linear.attn_qkv_proj.flatten_contiguous": { + "count": 50, + "mean_s": 4.358399964985438e-06, + "p50_s": 4.255999556335155e-06, + "p90_s": 4.454400368558709e-06, + "p95_s": 4.644000000553205e-06, + "p99_s": 6.431039619201322e-06, + "min_s": 4.079998689121567e-06, + "max_s": 7.951999577926472e-06 + }, + "linear.attn_qkv_proj.pre_quant_scale": { + "count": 50, + "mean_s": 0.0, + "p50_s": 0.0, + "p90_s": 0.0, + "p95_s": 0.0, + "p99_s": 0.0, + "min_s": 0.0, + "max_s": 0.0 + }, + "linear.attn_qkv_proj.packed_weight_wrapper": { + "count": 50, + "mean_s": 8.606720220996068e-06, + "p50_s": 8.119999620248564e-06, + "p90_s": 8.673600859765429e-06, + "p95_s": 9.055200825969221e-06, + "p99_s": 1.9636799952422696e-05, + "min_s": 7.58399983169511e-06, + "max_s": 2.927999958046712e-05 + }, + "linear.attn_qkv_proj.bias_cast": { + "count": 50, + "mean_s": 3.511040013108868e-06, + "p50_s": 3.4720005714916624e-06, + "p90_s": 3.604800258472096e-06, + "p95_s": 3.7135997445147947e-06, + "p99_s": 3.989119686593766e-06, + "min_s": 3.3760006772354245e-06, + "max_s": 4.239998816046864e-06 + }, + "linear.attn_qkv_proj.activation_scale": { + "count": 50, + "mean_s": 0.0017170804400302587, + "p50_s": 0.0017020014993249788, + "p90_s": 0.0017806290994485607, + "p95_s": 0.0017970364000575501, + "p99_s": 0.0018388336505449842, + "min_s": 0.0016812330013635801, + "max_s": 0.0018447210004524095 + }, + "linear.attn_qkv_proj.scale_to_device": { + "count": 50, + "mean_s": 1.0970560033456423e-05, + "p50_s": 9.424000381841324e-06, + "p90_s": 1.3892800961912145e-05, + "p95_s": 1.8935199113911944e-05, + "p99_s": 2.951103981104094e-05, + "min_s": 8.319999324157834e-06, + "max_s": 3.7695999708375894e-05 + }, + "linear.attn_qkv_proj.activation_quant_pack": { + "count": 50, + "mean_s": 0.0021769854400554324, + "p50_s": 0.0021467699989443645, + "p90_s": 0.0022839940002086223, + "p95_s": 0.0023105996001504536, + "p99_s": 0.002333132719668356, + "min_s": 0.002122096999300993, + "max_s": 0.0023364019998552976 + }, + "linear.attn_qkv_proj.activation_quant_wrap": { + "count": 50, + "mean_s": 6.238080022740177e-06, + "p50_s": 6.159999429655727e-06, + "p90_s": 6.5759997596615e-06, + "p95_s": 6.858400229248218e-06, + "p99_s": 7.811040286469504e-06, + "min_s": 5.727999450755306e-06, + "max_s": 8.352000804734416e-06 + }, + "linear.attn_qkv_proj.gemm": { + "count": 50, + "mean_s": 0.026236184619883716, + "p50_s": 0.026313476000723313, + "p90_s": 0.0267578761002369, + "p95_s": 0.0267910313997163, + "p99_s": 0.026870087120514655, + "min_s": 0.025206532998709008, + "max_s": 0.02692462999948475 + }, + "linear.attn_qkv_proj.slice_reshape": { + "count": 50, + "mean_s": 5.6448200120939875e-06, + "p50_s": 5.576001058216207e-06, + "p90_s": 5.824000254506245e-06, + "p95_s": 5.882399636902846e-06, + "p99_s": 7.303040310944193e-06, + "min_s": 5.296000381349586e-06, + "max_s": 8.431999958702363e-06 + }, + "attn_qkv_proj": { + "count": 50, + "mean_s": 0.030210273239936213, + "p50_s": 0.03035905599972466, + "p90_s": 0.030703277699649333, + "p95_s": 0.03072695820019362, + "p99_s": 0.030874537349391176, + "min_s": 0.029219621999800438, + "max_s": 0.03091569000025629 + }, + "attn_qkv_split_view": { + "count": 50, + "mean_s": 1.1833939897769596e-05, + "p50_s": 1.1328000255161896e-05, + "p90_s": 1.3222398774814793e-05, + "p95_s": 1.4168800498737254e-05, + "p99_s": 1.9776160061155663e-05, + "min_s": 1.046399847837165e-05, + "max_s": 2.4095999833662063e-05 + }, + "attn_qk_rms_rope": { + "count": 50, + "mean_s": 0.012088846719998401, + "p50_s": 0.012047434000123758, + "p90_s": 0.012417796300906048, + "p95_s": 0.012425689800329565, + "p99_s": 0.01257420811052725, + "min_s": 0.011746297001081984, + "max_s": 0.012706587000138825 + }, + "attention_kernel": { + "count": 50, + "mean_s": 0.26502367770001, + "p50_s": 0.265111950499886, + "p90_s": 0.2670563616989966, + "p95_s": 0.26712134680019517, + "p99_s": 0.2677792321504239, + "min_s": 0.262332772001173, + "max_s": 0.26781361299981654 + }, + "attn_output_reshape": { + "count": 50, + "mean_s": 4.21248005295638e-06, + "p50_s": 4.160000571573619e-06, + "p90_s": 4.355199598649051e-06, + "p95_s": 4.7128000915108714e-06, + "p99_s": 4.971359285264043e-06, + "min_s": 3.935998392989859e-06, + "max_s": 5.135998435434885e-06 + }, + "linear.attn_out_proj.flatten_contiguous": { + "count": 50, + "mean_s": 4.259839988662861e-06, + "p50_s": 4.175999492872506e-06, + "p90_s": 4.68319976789644e-06, + "p95_s": 4.8599995352560654e-06, + "p99_s": 5.284480539557989e-06, + "min_s": 3.952000042772852e-06, + "max_s": 5.504000000655651e-06 + }, + "linear.attn_out_proj.pre_quant_scale": { + "count": 50, + "mean_s": 0.0, + "p50_s": 0.0, + "p90_s": 0.0, + "p95_s": 0.0, + "p99_s": 0.0, + "min_s": 0.0, + "max_s": 0.0 + }, + "linear.attn_out_proj.packed_weight_wrapper": { + "count": 50, + "mean_s": 9.12031999177998e-06, + "p50_s": 8.799999704933725e-06, + "p90_s": 9.904000762617216e-06, + "p95_s": 1.0727198878157648e-05, + "p99_s": 1.329311962763313e-05, + "min_s": 8.143999366438948e-06, + "max_s": 1.3935999959358014e-05 + }, + "linear.attn_out_proj.bias_cast": { + "count": 50, + "mean_s": 3.559359975042753e-06, + "p50_s": 3.527999979269225e-06, + "p90_s": 3.7664001865778117e-06, + "p95_s": 3.832799484371207e-06, + "p99_s": 3.96991988964146e-06, + "min_s": 3.375998858246021e-06, + "max_s": 4.064000677317381e-06 + }, + "linear.attn_out_proj.activation_scale": { + "count": 50, + "mean_s": 0.0022237825599950157, + "p50_s": 0.0022224574986466905, + "p90_s": 0.0022319219999189953, + "p95_s": 0.0022335843996188487, + "p99_s": 0.0022542973603594876, + "min_s": 0.0022130420002213214, + "max_s": 0.0022664179996354505 + }, + "linear.attn_out_proj.scale_to_device": { + "count": 50, + "mean_s": 1.0166740030399524e-05, + "p50_s": 9.976000001188368e-06, + "p90_s": 1.0788800500449724e-05, + "p95_s": 1.1143999745399921e-05, + "p99_s": 1.3063200240139847e-05, + "min_s": 8.991999493446201e-06, + "max_s": 1.3807999494019896e-05 + }, + "linear.attn_out_proj.activation_quant_pack": { + "count": 50, + "mean_s": 0.0028620550799314513, + "p50_s": 0.0028830265000578947, + "p90_s": 0.0028994244999921647, + "p95_s": 0.002903895250892674, + "p99_s": 0.002973585690124309, + "min_s": 0.0028147700013505528, + "max_s": 0.0030201940007827943 + }, + "linear.attn_out_proj.activation_quant_wrap": { + "count": 50, + "mean_s": 6.35775981209008e-06, + "p50_s": 6.296000719885342e-06, + "p90_s": 6.663998829026241e-06, + "p95_s": 6.848000339232385e-06, + "p99_s": 7.630399486515668e-06, + "min_s": 5.935999070061371e-06, + "max_s": 8.335999154951423e-06 + }, + "linear.attn_out_proj.gemm": { + "count": 50, + "mean_s": 0.008039239360005012, + "p50_s": 0.008028734499930579, + "p90_s": 0.008143914901120297, + "p95_s": 0.008157982450484269, + "p99_s": 0.00827479994910391, + "min_s": 0.007810614000845817, + "max_s": 0.008372870999664883 + }, + "linear.attn_out_proj.slice_reshape": { + "count": 50, + "mean_s": 5.5884800531202926e-06, + "p50_s": 5.583999154623598e-06, + "p90_s": 5.7791990911937315e-06, + "p95_s": 5.81680033064913e-06, + "p99_s": 5.995040301058907e-06, + "min_s": 5.312000212143175e-06, + "max_s": 6.144000508356839e-06 + }, + "attn_out_proj": { + "count": 50, + "mean_s": 0.013204702960210852, + "p50_s": 0.013201394999668992, + "p90_s": 0.013340711800265127, + "p95_s": 0.01335459340034504, + "p99_s": 0.01355505611976696, + "min_s": 0.012927738000144018, + "max_s": 0.013737978999415645 + }, + "gate_msa": { + "count": 50, + "mean_s": 0.008690837319918501, + "p50_s": 0.008646599499115837, + "p90_s": 0.008749174999866226, + "p95_s": 0.009079649749855888, + "p99_s": 0.009193422359730903, + "min_s": 0.008623383000667673, + "max_s": 0.00926199099922087 + }, + "norm2": { + "count": 50, + "mean_s": 0.003568036460201256, + "p50_s": 0.003531907000251522, + "p90_s": 0.0036303837016021133, + "p95_s": 0.003743043001395563, + "p99_s": 0.0038393764505599394, + "min_s": 0.003515746999255498, + "max_s": 0.0038490269998874282 + }, + "modulate_mlp": { + "count": 50, + "mean_s": 0.010634301519858128, + "p50_s": 0.010705913000492728, + "p90_s": 0.010969404899333313, + "p95_s": 0.01097855734933546, + "p99_s": 0.01102472357910301, + "min_s": 0.010306168000170146, + "max_s": 0.011047977999623981 + }, + "linear.mlp_fc1.flatten_contiguous": { + "count": 50, + "mean_s": 4.33599987445632e-06, + "p50_s": 4.303999958210625e-06, + "p90_s": 4.4959997467231005e-06, + "p95_s": 4.527999408310279e-06, + "p99_s": 5.126080432091839e-06, + "min_s": 4.064000677317381e-06, + "max_s": 5.424000846687704e-06 + }, + "linear.mlp_fc1.pre_quant_scale": { + "count": 50, + "mean_s": 0.0, + "p50_s": 0.0, + "p90_s": 0.0, + "p95_s": 0.0, + "p99_s": 0.0, + "min_s": 0.0, + "max_s": 0.0 + }, + "linear.mlp_fc1.packed_weight_wrapper": { + "count": 50, + "mean_s": 8.286079973913729e-06, + "p50_s": 8.199999683711212e-06, + "p90_s": 8.604798495071009e-06, + "p95_s": 8.881600660970434e-06, + "p99_s": 1.04201599060616e-05, + "min_s": 7.728000127826817e-06, + "max_s": 1.062400042428635e-05 + }, + "linear.mlp_fc1.bias_cast": { + "count": 50, + "mean_s": 3.53920000634389e-06, + "p50_s": 3.48799949279055e-06, + "p90_s": 3.633599772001617e-06, + "p95_s": 3.6976009141653774e-06, + "p99_s": 4.592639797920124e-06, + "min_s": 3.3599990274524316e-06, + "max_s": 5.4079991969047114e-06 + }, + "linear.mlp_fc1.activation_scale": { + "count": 50, + "mean_s": 0.0017303090999121196, + "p50_s": 0.0017195689997606678, + "p90_s": 0.0018050011001832901, + "p95_s": 0.0018161890008741464, + "p99_s": 0.0018366168388274672, + "min_s": 0.0016687049992469838, + "max_s": 0.0018491529990569688 + }, + "linear.mlp_fc1.scale_to_device": { + "count": 50, + "mean_s": 1.0146900021936745e-05, + "p50_s": 9.456000043428503e-06, + "p90_s": 1.0617600673867857e-05, + "p95_s": 1.2792001143679946e-05, + "p99_s": 2.6764480571728167e-05, + "min_s": 8.479999451083131e-06, + "max_s": 3.44320014846744e-05 + }, + "linear.mlp_fc1.activation_quant_pack": { + "count": 50, + "mean_s": 0.0022118504599347943, + "p50_s": 0.0022029219999240013, + "p90_s": 0.0022765955998693245, + "p95_s": 0.0023190750502180887, + "p99_s": 0.00235583480061905, + "min_s": 0.0021436339993670117, + "max_s": 0.0023676340006204555 + }, + "linear.mlp_fc1.activation_quant_wrap": { + "count": 50, + "mean_s": 6.254719919525087e-06, + "p50_s": 6.223999662324786e-06, + "p90_s": 6.531200051540508e-06, + "p95_s": 6.748800569766899e-06, + "p99_s": 7.058879964461084e-06, + "min_s": 5.840000085299835e-06, + "max_s": 7.200000254670158e-06 + }, + "linear.mlp_fc1.gemm": { + "count": 50, + "mean_s": 0.03515858558006585, + "p50_s": 0.03511626750059804, + "p90_s": 0.03549363759975677, + "p95_s": 0.03557769874960286, + "p99_s": 0.03572802138010957, + "min_s": 0.03414487400004873, + "max_s": 0.03578301800007466 + }, + "linear.mlp_fc1.slice_reshape": { + "count": 50, + "mean_s": 5.572800073423423e-06, + "p50_s": 5.551999493036419e-06, + "p90_s": 5.761599277320784e-06, + "p95_s": 5.841600068379194e-06, + "p99_s": 6.067840458854333e-06, + "min_s": 5.343999873730354e-06, + "max_s": 6.2560011429013684e-06 + }, + "mlp_fc1": { + "count": 50, + "mean_s": 0.03918287416006933, + "p50_s": 0.03917559199999232, + "p90_s": 0.039542551699923933, + "p95_s": 0.039647989400327786, + "p99_s": 0.039722929230501906, + "min_s": 0.038156013000843814, + "max_s": 0.03972972700103128 + }, + "mlp_swiglu": { + "count": 50, + "mean_s": 0.024074891440068313, + "p50_s": 0.024049810999713372, + "p90_s": 0.024356551100390787, + "p95_s": 0.024452045749512763, + "p99_s": 0.024497954870494137, + "min_s": 0.023798129999704543, + "max_s": 0.024512820000381907 + }, + "linear.mlp_fc2.flatten_contiguous": { + "count": 50, + "mean_s": 4.211519808450248e-06, + "p50_s": 4.167999577475712e-06, + "p90_s": 4.403200546221342e-06, + "p95_s": 4.511200586421182e-06, + "p99_s": 4.908320879621896e-06, + "min_s": 3.9360002119792625e-06, + "max_s": 5.1200004236306995e-06 + }, + "linear.mlp_fc2.pre_quant_scale": { + "count": 50, + "mean_s": 0.0, + "p50_s": 0.0, + "p90_s": 0.0, + "p95_s": 0.0, + "p99_s": 0.0, + "min_s": 0.0, + "max_s": 0.0 + }, + "linear.mlp_fc2.packed_weight_wrapper": { + "count": 50, + "mean_s": 7.91585993283661e-06, + "p50_s": 7.840000762371346e-06, + "p90_s": 8.257599984062835e-06, + "p95_s": 8.703200273885157e-06, + "p99_s": 9.320799654233269e-06, + "min_s": 7.296001058421098e-06, + "max_s": 9.359999239677563e-06 + }, + "linear.mlp_fc2.bias_cast": { + "count": 50, + "mean_s": 3.5664000824908725e-06, + "p50_s": 3.5360008041607216e-06, + "p90_s": 3.684801049530506e-06, + "p95_s": 3.763200220419094e-06, + "p99_s": 3.9768007081875104e-06, + "min_s": 3.3760006772354245e-06, + "max_s": 4.016001184936613e-06 + }, + "linear.mlp_fc2.activation_scale": { + "count": 50, + "mean_s": 0.004592374580024625, + "p50_s": 0.004559275500469084, + "p90_s": 0.004735260000234121, + "p95_s": 0.004762049599139573, + "p99_s": 0.004766211350379308, + "min_s": 0.004506786999627366, + "max_s": 0.00476659600099083 + }, + "linear.mlp_fc2.scale_to_device": { + "count": 50, + "mean_s": 1.0288320081599523e-05, + "p50_s": 9.424000381841324e-06, + "p90_s": 1.2665600115724375e-05, + "p95_s": 1.4833600562269567e-05, + "p99_s": 2.0252960439393047e-05, + "min_s": 8.70400072017219e-06, + "max_s": 2.4416000087512657e-05 + }, + "linear.mlp_fc2.activation_quant_pack": { + "count": 50, + "mean_s": 0.005908733240030415, + "p50_s": 0.005951356499281246, + "p90_s": 0.006028886500280351, + "p95_s": 0.0060370161999344415, + "p99_s": 0.0060677086695250185, + "min_s": 0.00564893199953076, + "max_s": 0.0060685009993903805 + }, + "linear.mlp_fc2.activation_quant_wrap": { + "count": 50, + "mean_s": 6.152319983812049e-06, + "p50_s": 6.039999789209105e-06, + "p90_s": 6.579199725820218e-06, + "p95_s": 6.739199670846574e-06, + "p99_s": 7.526239696744594e-06, + "min_s": 5.711999619961716e-06, + "max_s": 8.224000339396298e-06 + }, + "linear.mlp_fc2.gemm": { + "count": 50, + "mean_s": 0.05473969570000918, + "p50_s": 0.054685164999682456, + "p90_s": 0.05519963150054536, + "p95_s": 0.05526944609919156, + "p99_s": 0.05548157012059164, + "min_s": 0.05419282500042755, + "max_s": 0.05548338900007366 + }, + "linear.mlp_fc2.slice_reshape": { + "count": 50, + "mean_s": 5.6211200353573075e-06, + "p50_s": 5.527999746846035e-06, + "p90_s": 5.95199890085496e-06, + "p95_s": 6.123199909779941e-06, + "p99_s": 6.801280833315104e-06, + "min_s": 5.2320010581752285e-06, + "max_s": 6.8640001700259745e-06 + }, + "mlp_fc2": { + "count": 50, + "mean_s": 0.06531846147994656, + "p50_s": 0.06529986749956151, + "p90_s": 0.06578957450019515, + "p95_s": 0.06586715780076702, + "p99_s": 0.06601896072916133, + "min_s": 0.0645643679999921, + "max_s": 0.06614333399920724 + }, + "gate_mlp": { + "count": 50, + "mean_s": 0.008981906899825844, + "p50_s": 0.0086321990002034, + "p90_s": 0.009006409600442567, + "p95_s": 0.009076948600159084, + "p99_s": 0.01610825177995136, + "min_s": 0.008592614998633508, + "max_s": 0.022754208999685943 + }, + "block_total": { + "count": 50, + "mean_s": 0.4951837136800532, + "p50_s": 0.4949583219995475, + "p90_s": 0.49724229510084117, + "p95_s": 0.497455911299312, + "p99_s": 0.502053749699644, + "min_s": 0.49234532799891895, + "max_s": 0.5055920739996509 + } + }, + "module_forward": { + "count": 50, + "mean_s": 0.42752286658007504, + "p50_s": 0.42741008050052187, + "p90_s": 0.42897752329972716, + "p95_s": 0.42947807479968103, + "p99_s": 0.43056982682959644, + "min_s": 0.4248241990007955, + "max_s": 0.4311483860001317 + }, + "module_forward_checksum": -15117459456.0, + "fused_elementwise": true, + "profiler_summary": { + "profiled_iterations": 1, + "runtime_kernel_launches": 0, + "runtime_kernel_launches_per_block": 0.0, + "positive_self_device_allocated_bytes": 8009579008 + }, + "profiler_top_events": [ + { + "key": "aten::rms_norm", + "count": 2, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 1546.5930000000008, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": -303104, + "input_shapes": "[[37810, 5376], [], [5376], []]" + }, + { + "key": "aten::_fused_rms_norm", + "count": 2, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 1535.761, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37810, 5376], [], [5376], []]" + }, + { + "key": "aten::linear", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 1039.4409999999993, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37810, 5376], [21504, 5376], []]" + }, + { + "key": "comfy_kitchen::scaled_mm_nvfp4", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 822.1930000000002, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": -512, + "input_shapes": "[[37824, 2688], [21504, 2688], [], [], [37888, 336], [21504, 336], [], [], []]" + }, + { + "key": "comfy_kitchen::rms_rope_split_half_", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 339.29699999999957, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[1, 37810, 56, 128], [1, 37810, 56, 128], [1, 37810, 1, 48, 2, 2], [128], [128], [], []]" + }, + { + "key": "aten::contiguous", + "count": 4, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 195.40800000000013, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[3, 5376], []]" + }, + { + "key": "aten::clone", + "count": 4, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 185.59999999999877, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[3, 5376], []]" + }, + { + "key": "aten::zeros", + "count": 4, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 155.15199999999913, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[], [], [], [], []]" + }, + { + "key": "aten::empty", + "count": 43, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 138.87999999999772, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 8009561088, + "input_shapes": "[[], [], [], [], [], []]" + }, + { + "key": "aten::linear", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 136.91200000000026, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37810, 7168], [5376, 7168], []]" + }, + { + "key": "aten::copy_", + "count": 4, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 132.60799999999972, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[3, 5376], [3, 5376], []]" + }, + { + "key": "comfy_kitchen::scaled_mm_nvfp4", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 118.55999999999949, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": -512, + "input_shapes": "[[37824, 3584], [5376, 3584], [], [], [37888, 448], [5376, 448], [], [], []]" + }, + { + "key": "aten::zero_", + "count": 2, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 104.68799999999919, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37888, 336]]" + }, + { + "key": "comfy_kitchen::quantize_nvfp4", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 101.3119999999999, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37810, 7168], [], [], [], []]" + }, + { + "key": "aten::mean", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 95.15200000000004, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 14336, + "input_shapes": "[[1, 37810, 56, 128], [], [], []]" + }, + { + "key": "sageattention_sm89::qk_int8_sv_f8_accum_f16_fuse_v_scale_attn_inst_buf", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 94.65599999999995, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[1, 37810, 56, 128], [1, 37810, 56, 128], [1, 128, 56, 37824], [1, 37810, 56, 128], [1, 56, 1184], [1, 56, 591], [1, 56, 128], [], [], [], [], []]" + }, + { + "key": "aten::linear", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 92.51199999999972, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37810, 5376], [28672, 5376], []]" + }, + { + "key": "comfy_kitchen::scaled_mm_nvfp4", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 78.20799999999963, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": -512, + "input_shapes": "[[37824, 2688], [28672, 2688], [], [], [37888, 336], [28672, 336], [], [], []]" + }, + { + "key": "aten::mul", + "count": 4, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 71.6319999999996, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 2048, + "input_shapes": "[[], []]" + }, + { + "key": "aten::_to_copy", + "count": 2, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 59.3769999999995, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[], [], [], [], [], [], []]" + }, + { + "key": "aten::to", + "count": 9, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 46.91300000000001, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[], [], [], [], []]" + }, + { + "key": "aten::fill_", + "count": 2, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 44.735999999999876, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37888, 336], []]" + }, + { + "key": "aten::copy_", + "count": 2, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 40.20899999999983, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[], [], []]" + }, + { + "key": "aten::split", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 32.384000000000015, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37810, 21504], [], []]" + }, + { + "key": "aten::empty_like", + "count": 4, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 26.607999999999493, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[3, 5376], [], [], [], [], []]" + }, + { + "key": "aten::to", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 24.047999999999774, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[], [], [], [], [], []]" + }, + { + "key": "aten::narrow", + "count": 3, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 18.367999999999483, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37810, 21504], [], [], []]" + }, + { + "key": "aten::zeros_like", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 16.20800000000054, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[1], [], [], [], [], []]" + }, + { + "key": "aten::reshape", + "count": 5, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 16.112000000000535, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[], []]" + }, + { + "key": "aten::slice", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 16.096000000000004, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37824, 21504], [], [], [], []]" + }, + { + "key": "aten::empty_strided", + "count": 3, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 11.568000000000211, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 1536, + "input_shapes": "[[], [], [], [], [], []]" + }, + { + "key": "aten::zero_", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 10.78399999999965, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[1]]" + }, + { + "key": "aten::fill_", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 9.583999999999833, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[1], []]" + }, + { + "key": "aten::view", + "count": 4, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 9.488000000000284, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37888, 336], []]" + }, + { + "key": "aten::zero_", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 8.943999999999505, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37888, 448]]" + }, + { + "key": "aten::squeeze", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 7.632000000000517, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[1, 1, 56, 128], []]" + }, + { + "key": "aten::view", + "count": 5, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 7.376000000002023, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[], []]" + }, + { + "key": "aten::fill_", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 6.943999999999505, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37888, 448], []]" + }, + { + "key": "aten::slice", + "count": 3, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 6.752000000001317, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37810, 21504], [], [], [], []]" + }, + { + "key": "aten::zero_", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 6.496000000000095, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37888, 896]]" + }, + { + "key": "aten::to", + "count": 2, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 5.280000000000399, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[5376], [], [], [], []]" + }, + { + "key": "aten::fill_", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 5.168000000000575, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37888, 896], []]" + }, + { + "key": "aten::as_strided", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 5.023999999999887, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37824, 21504], [], [], []]" + }, + { + "key": "aten::view", + "count": 2, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 4.879999999999882, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37810], []]" + }, + { + "key": "aten::slice", + "count": 2, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 3.8400000000001455, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37824, 5376], [], [], [], []]" + }, + { + "key": "aten::view", + "count": 4, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 3.7440000000005966, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37810, 7168], []]" + }, + { + "key": "aten::empty_like", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 3.344000000000051, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[1], [], [], [], [], []]" + }, + { + "key": "aten::as_strided", + "count": 3, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 2.256000000000313, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37810, 21504], [], [], []]" + }, + { + "key": "aten::reshape", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 2.207999999999629, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37810, 7168], []]" + }, + { + "key": "aten::alias", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 2.1279999999997017, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37810, 21504]]" + }, + { + "key": "aten::slice", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 2.0, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37824, 28672], [], [], [], []]" + }, + { + "key": "aten::view", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 1.9520000000002256, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[21504, 336], []]" + }, + { + "key": "aten::as_strided", + "count": 4, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 1.88799999999992, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[1, 37810, 56, 128], [], [], []]" + }, + { + "key": "aten::reshape", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 1.8079999999999927, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[1, 37810, 56, 128], []]" + }, + { + "key": "aten::reshape", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 1.199999999999818, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37810, 5376], []]" + }, + { + "key": "aten::as_strided", + "count": 2, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 0.9599999999991269, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37824, 5376], [], [], []]" + }, + { + "key": "aten::alias", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 0.8640000000004875, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[1, 37810, 56, 128]]" + }, + { + "key": "aten::view", + "count": 2, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 0.7359999999998763, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37888, 448], []]" + }, + { + "key": "aten::as_strided", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 0.4960000000000946, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37824, 28672], [], [], []]" + }, + { + "key": "aten::view", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 0.4320000000006985, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[28672, 336], []]" + }, + { + "key": "aten::view", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 0.41600000000016735, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[1, 37810, 56, 128], []]" + }, + { + "key": "aten::as_strided", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 0.3999999999996362, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[1, 1, 56, 128], [], [], []]" + }, + { + "key": "aten::view", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 0.3999999999996362, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[5376, 448], []]" + }, + { + "key": "aten::alias", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 0.3999999999996362, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37810, 5376]]" + }, + { + "key": "aten::view", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 0.38400000000001455, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37810, 5376], []]" + }, + { + "key": "aten::alias", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 0.35199999999986176, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37810, 28672]]" + }, + { + "key": "aten::view", + "count": 1, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 0.3040000000000873, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": 0, + "input_shapes": "[[37888, 896], []]" + }, + { + "key": "[memory]", + "count": 38, + "device_type": "DeviceType.CPU", + "cpu_time_total_us": 0.0, + "device_time_total_us": 0.0, + "self_device_time_total_us": 0.0, + "self_device_memory_usage_bytes": -8009274368, + "input_shapes": "[]" + } + ] +} \ No newline at end of file diff --git a/benchmarks/gb10-post-fc2-block24-targeted-20260826.csv b/benchmarks/gb10-post-fc2-block24-targeted-20260826.csv new file mode 100644 index 0000000..557838e --- /dev/null +++ b/benchmarks/gb10-post-fc2-block24-targeted-20260826.csv @@ -0,0 +1,7 @@ +"ID","Process ID","Process Name","Host Name","Kernel Name","Context","Stream","Block Size","Grid Size","Device","CC","c2clink__enabled_mask","c2clink__present","derived__local_spilling_requests","derived__local_spilling_requests_pct","derived__pct_occupancy_per_barrier_count","derived__pct_occupancy_per_block_size","derived__pct_occupancy_per_register_count","derived__pct_occupancy_per_shared_mem_size","derived_tempMetric0","device__attribute_architecture","device__attribute_async_engine_count","device__attribute_can_flush_remote_writes","device__attribute_can_map_host_memory","device__attribute_can_tex2d_gather","device__attribute_can_use_64_bit_stream_mem_ops","device__attribute_can_use_64_bit_stream_mem_ops_v1","device__attribute_can_use_host_pointer_for_registered_mem","device__attribute_can_use_stream_mem_ops_v1","device__attribute_can_use_stream_wait_value_nor","device__attribute_can_use_stream_wait_value_nor_v1","device__attribute_chip","device__attribute_clock_rate","device__attribute_cluster_launch","device__attribute_compute_capability_major","device__attribute_compute_capability_minor","device__attribute_compute_mode","device__attribute_compute_preemption_supported","device__attribute_concurrent_kernels","device__attribute_concurrent_managed_access","device__attribute_confidential_computing_mode","device__attribute_cooperative_launch","device__attribute_cooperative_multi_device_launch","device__attribute_deferred_mapping_cuda_array_supported","device__attribute_device_index","device__attribute_direct_managed_mem_access_from_host","device__attribute_display_name","device__attribute_dma_buf_supported","device__attribute_ecc_enabled","device__attribute_fb_bus_width","device__attribute_fbp_count","device__attribute_generic_compression_supported","device__attribute_global_l1_cache_supported","device__attribute_global_memory_bus_width","device__attribute_gpu_direct_rdma_flush_writes_options","device__attribute_gpu_direct_rdma_supported","device__attribute_gpu_direct_rdma_with_cuda_vmm_supported","device__attribute_gpu_direct_rdma_writes_ordering","device__attribute_gpu_overlap","device__attribute_gpu_pci_device_id","device__attribute_gpu_pci_ext_device_id","device__attribute_gpu_pci_ext_downstream_link_rate","device__attribute_gpu_pci_ext_downstream_link_width","device__attribute_gpu_pci_ext_gen","device__attribute_gpu_pci_ext_gpu_gen","device__attribute_gpu_pci_ext_gpu_link_rate","device__attribute_gpu_pci_ext_gpu_link_width","device__attribute_gpu_pci_revision_id","device__attribute_gpu_pci_sub_system_id","device__attribute_handle_type_fabric_supported","device__attribute_handle_type_posix_file_descriptor_supported","device__attribute_handle_type_win32_handle_supported","device__attribute_handle_type_win32_kmt_handle_supported","device__attribute_host_native_atomic_supported","device__attribute_host_numa_id","device__attribute_host_register_supported","device__attribute_implementation","device__attribute_integrated","device__attribute_ipc_event_supported","device__attribute_kernel_exec_timeout","device__attribute_l2_cache_size","device__attribute_l2s_count","device__attribute_limits_max_cta_per_sm","device__attribute_limits_num_tpcs","device__attribute_local_l1_cache_supported","device__attribute_managed_memory","device__attribute_max_access_policy_window_size","device__attribute_max_block_dim_x","device__attribute_max_block_dim_y","device__attribute_max_block_dim_z","device__attribute_max_blocks_per_multiprocessor","device__attribute_max_gpu_frequency_khz","device__attribute_max_grid_dim_x","device__attribute_max_grid_dim_y","device__attribute_max_grid_dim_z","device__attribute_max_ipc_per_multiprocessor","device__attribute_max_ipc_per_scheduler","device__attribute_max_mem_frequency_khz","device__attribute_max_persisting_l2_cache_size","device__attribute_max_pitch","device__attribute_max_registers_per_block","device__attribute_max_registers_per_multiprocessor","device__attribute_max_registers_per_thread","device__attribute_max_shared_memory_per_block","device__attribute_max_shared_memory_per_block_optin","device__attribute_max_shared_memory_per_multiprocessor","device__attribute_max_threads_per_block","device__attribute_max_threads_per_multiprocessor","device__attribute_max_warps_per_multiprocessor","device__attribute_max_warps_per_scheduler","device__attribute_maximum_surface1d_layered_layers","device__attribute_maximum_surface1d_layered_width","device__attribute_maximum_surface1d_width","device__attribute_maximum_surface2d_height","device__attribute_maximum_surface2d_layered_height","device__attribute_maximum_surface2d_layered_layers","device__attribute_maximum_surface2d_layered_width","device__attribute_maximum_surface2d_width","device__attribute_maximum_surface3d_depth","device__attribute_maximum_surface3d_height","device__attribute_maximum_surface3d_width","device__attribute_maximum_surfacecubemap_layered_layers","device__attribute_maximum_surfacecubemap_layered_width","device__attribute_maximum_surfacecubemap_width","device__attribute_maximum_texture1d_layered_layers","device__attribute_maximum_texture1d_layered_width","device__attribute_maximum_texture1d_linear_width","device__attribute_maximum_texture1d_mipmapped_width","device__attribute_maximum_texture1d_width","device__attribute_maximum_texture2d_gather_height","device__attribute_maximum_texture2d_gather_width","device__attribute_maximum_texture2d_height","device__attribute_maximum_texture2d_layered_height","device__attribute_maximum_texture2d_layered_layers","device__attribute_maximum_texture2d_layered_width","device__attribute_maximum_texture2d_linear_height","device__attribute_maximum_texture2d_linear_pitch","device__attribute_maximum_texture2d_linear_width","device__attribute_maximum_texture2d_mipmapped_height","device__attribute_maximum_texture2d_mipmapped_width","device__attribute_maximum_texture2d_width","device__attribute_maximum_texture3d_depth","device__attribute_maximum_texture3d_depth_alternate","device__attribute_maximum_texture3d_height","device__attribute_maximum_texture3d_height_alternate","device__attribute_maximum_texture3d_width","device__attribute_maximum_texture3d_width_alternate","device__attribute_maximum_texturecubemap_layered_layers","device__attribute_maximum_texturecubemap_layered_width","device__attribute_maximum_texturecubemap_width","device__attribute_mem_sync_domain_count","device__attribute_memory_clock_rate","device__attribute_memory_pools_supported","device__attribute_mempool_supported_handle_types","device__attribute_mps_enabled","device__attribute_multi_gpu_board","device__attribute_multi_gpu_board_group_id","device__attribute_multicast_supported","device__attribute_multiprocessor_count","device__attribute_num_l2s_per_fbp","device__attribute_num_schedulers_per_multiprocessor","device__attribute_num_tex_per_multiprocessor","device__attribute_numa_config","device__attribute_pageable_memory_access","device__attribute_pageable_memory_access_uses_host_page_tables","device__attribute_pci_bus_id","device__attribute_pci_device_id","device__attribute_pci_domain_id","device__attribute_ram_location","device__attribute_ram_type","device__attribute_reserved_shared_memory_per_block","device__attribute_sass_level","device__attribute_single_to_double_precision_perf_ratio","device__attribute_sparse_cuda_array_supported","device__attribute_stream_priorities_supported","device__attribute_surface_alignment","device__attribute_tcc_driver","device__attribute_tensor_map_access_supported","device__attribute_texture_alignment","device__attribute_texture_pitch_alignment","device__attribute_total_constant_memory","device__attribute_total_memory","device__attribute_unified_addressing","device__attribute_unified_function_pointers","device__attribute_virtual_address_management_supported","device__attribute_warp_size","gpc__cycles_elapsed.avg","gpc__cycles_elapsed.avg.per_second","gpc__cycles_elapsed.max","gpc__cycles_elapsed.max.per_second","gpc__cycles_elapsed.min","gpc__cycles_elapsed.min.per_second","gpc__cycles_elapsed.sum","gpc__cycles_elapsed.sum.per_second","gpu__compute_memory_access_throughput.avg.pct_of_peak_sustained_elapsed","gpu__compute_memory_access_throughput.max.pct_of_peak_sustained_elapsed","gpu__compute_memory_access_throughput.min.pct_of_peak_sustained_elapsed","gpu__compute_memory_access_throughput.sum.pct_of_peak_sustained_elapsed","gpu__compute_memory_access_throughput_internal_activity.avg.pct_of_peak_sustained_elapsed","gpu__compute_memory_access_throughput_internal_activity.max.pct_of_peak_sustained_elapsed","gpu__compute_memory_access_throughput_internal_activity.min.pct_of_peak_sustained_elapsed","gpu__compute_memory_access_throughput_internal_activity.sum.pct_of_peak_sustained_elapsed","gpu__compute_memory_request_throughput.avg.pct_of_peak_sustained_elapsed","gpu__compute_memory_request_throughput.max.pct_of_peak_sustained_elapsed","gpu__compute_memory_request_throughput.min.pct_of_peak_sustained_elapsed","gpu__compute_memory_request_throughput.sum.pct_of_peak_sustained_elapsed","gpu__compute_memory_request_throughput_internal_activity.avg.pct_of_peak_sustained_elapsed","gpu__compute_memory_request_throughput_internal_activity.max.pct_of_peak_sustained_elapsed","gpu__compute_memory_request_throughput_internal_activity.min.pct_of_peak_sustained_elapsed","gpu__compute_memory_request_throughput_internal_activity.sum.pct_of_peak_sustained_elapsed","gpu__compute_memory_throughput.avg.pct_of_peak_sustained_elapsed","gpu__compute_memory_throughput.max.pct_of_peak_sustained_elapsed","gpu__compute_memory_throughput.min.pct_of_peak_sustained_elapsed","gpu__compute_memory_throughput.sum.pct_of_peak_sustained_elapsed","gpu__time_duration.avg","gpu__time_duration.max","gpu__time_duration.min","gpu__time_duration.sum","gr__workids_granted.avg","gr__workids_granted.max","gr__workids_granted.min","gr__workids_granted.sum","gr__workids_granted_as_ctas.avg","gr__workids_granted_as_ctas.max","gr__workids_granted_as_ctas.min","gr__workids_granted_as_ctas.sum","gr__workids_requested.avg","gr__workids_requested.max","gr__workids_requested.min","gr__workids_requested.sum","idc__request_cycles_active.avg.pct_of_peak_sustained_elapsed","idc__request_cycles_active.max.pct_of_peak_sustained_elapsed","idc__request_cycles_active.min.pct_of_peak_sustained_elapsed","idc__request_cycles_active.sum.pct_of_peak_sustained_elapsed","inst_executed","l1tex__data_bank_reads.avg.pct_of_peak_sustained_elapsed","l1tex__data_bank_reads.max.pct_of_peak_sustained_elapsed","l1tex__data_bank_reads.min.pct_of_peak_sustained_elapsed","l1tex__data_bank_reads.sum.pct_of_peak_sustained_elapsed","l1tex__data_bank_writes.avg.pct_of_peak_sustained_elapsed","l1tex__data_bank_writes.max.pct_of_peak_sustained_elapsed","l1tex__data_bank_writes.min.pct_of_peak_sustained_elapsed","l1tex__data_bank_writes.sum.pct_of_peak_sustained_elapsed","l1tex__data_pipe_lsu_wavefronts.avg.pct_of_peak_sustained_elapsed","l1tex__data_pipe_lsu_wavefronts.max.pct_of_peak_sustained_elapsed","l1tex__data_pipe_lsu_wavefronts.min.pct_of_peak_sustained_elapsed","l1tex__data_pipe_lsu_wavefronts.sum.pct_of_peak_sustained_elapsed","l1tex__data_pipe_tex_wavefronts.avg.pct_of_peak_sustained_elapsed","l1tex__data_pipe_tex_wavefronts.max.pct_of_peak_sustained_elapsed","l1tex__data_pipe_tex_wavefronts.min.pct_of_peak_sustained_elapsed","l1tex__data_pipe_tex_wavefronts.sum.pct_of_peak_sustained_elapsed","l1tex__f_wavefronts.avg.pct_of_peak_sustained_elapsed","l1tex__f_wavefronts.max.pct_of_peak_sustained_elapsed","l1tex__f_wavefronts.min.pct_of_peak_sustained_elapsed","l1tex__f_wavefronts.sum.pct_of_peak_sustained_elapsed","l1tex__lsu_writeback_active.avg.pct_of_peak_sustained_elapsed","l1tex__lsu_writeback_active.max.pct_of_peak_sustained_elapsed","l1tex__lsu_writeback_active.min.pct_of_peak_sustained_elapsed","l1tex__lsu_writeback_active.sum.pct_of_peak_sustained_elapsed","l1tex__lsuin_requests.avg.pct_of_peak_sustained_elapsed","l1tex__lsuin_requests.max.pct_of_peak_sustained_elapsed","l1tex__lsuin_requests.min.pct_of_peak_sustained_elapsed","l1tex__lsuin_requests.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_req_cycles_active.avg.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_req_cycles_active.max.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_req_cycles_active.min.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_req_cycles_active.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors.avg.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors.max.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors.min.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors.sum.pct_of_peak_sustained_elapsed","l1tex__t_sector_hit_rate.pct","l1tex__tex_writeback_active.avg.pct_of_peak_sustained_elapsed","l1tex__tex_writeback_active.max.pct_of_peak_sustained_elapsed","l1tex__tex_writeback_active.min.pct_of_peak_sustained_elapsed","l1tex__tex_writeback_active.sum.pct_of_peak_sustained_elapsed","l1tex__texin_sm2tex_req_cycles_active.avg.pct_of_peak_sustained_elapsed","l1tex__texin_sm2tex_req_cycles_active.max.pct_of_peak_sustained_elapsed","l1tex__texin_sm2tex_req_cycles_active.min.pct_of_peak_sustained_elapsed","l1tex__texin_sm2tex_req_cycles_active.sum.pct_of_peak_sustained_elapsed","l1tex__throughput.avg.pct_of_peak_sustained_active","l1tex__throughput.max.pct_of_peak_sustained_active","l1tex__throughput.min.pct_of_peak_sustained_active","l1tex__throughput.sum.pct_of_peak_sustained_active","launch__barrier_count","launch__block_dim_x","launch__block_dim_y","launch__block_dim_z","launch__block_size","launch__cluster_dim_x","launch__cluster_dim_y","launch__cluster_dim_z","launch__cluster_max_active","launch__cluster_max_potential_size","launch__cluster_scheduling_policy","launch__cluster_size","launch__context_id","launch__device_id","launch__func_cache_config","launch__function_pcs","launch__grid_dim_x","launch__grid_dim_y","launch__grid_dim_z","launch__grid_size","launch__kernel_name","launch__occupancy_cluster_gpu_pct","launch__occupancy_cluster_pct","launch__occupancy_limit_barriers","launch__occupancy_limit_blocks","launch__occupancy_limit_registers","launch__occupancy_limit_shared_mem","launch__occupancy_limit_warps","launch__occupancy_per_barrier_count","launch__occupancy_per_block_size","launch__occupancy_per_cluster_size","launch__occupancy_per_register_count","launch__occupancy_per_shared_mem_size","launch__persisting_l2_cache_size","launch__preferred_cluster_dim_x","launch__preferred_cluster_dim_y","launch__preferred_cluster_dim_z","launch__preferred_cluster_size","launch__registers_per_thread","launch__registers_per_thread_allocated","launch__shared_mem_config_size","launch__shared_mem_per_block","launch__shared_mem_per_block_allocated","launch__shared_mem_per_block_driver","launch__shared_mem_per_block_dynamic","launch__shared_mem_per_block_static","launch__sm_count","launch__stack_size","launch__stream_id","launch__thread_count","launch__tpc_count","launch__tpc_enabled","launch__uses_cdp","launch__uses_green_context","launch__uses_mps","launch__uses_nvlink_centric_scheduling","launch__uses_vgpu","launch__waves_per_multiprocessor","lts__average_gcomp_input_sector_success_rate.pct","lts__average_gcomp_output_sector_compression_achieved_rate.ratio","lts__d_atomic_input_cycles_active.avg.pct_of_peak_sustained_elapsed","lts__d_atomic_input_cycles_active.max.pct_of_peak_sustained_elapsed","lts__d_atomic_input_cycles_active.min.pct_of_peak_sustained_elapsed","lts__d_atomic_input_cycles_active.sum.pct_of_peak_sustained_elapsed","lts__d_decomp_input_sectors.avg.pct_of_peak_sustained_elapsed","lts__d_decomp_input_sectors.max.pct_of_peak_sustained_elapsed","lts__d_decomp_input_sectors.min.pct_of_peak_sustained_elapsed","lts__d_decomp_input_sectors.sum.pct_of_peak_sustained_elapsed","lts__d_sectors.avg.pct_of_peak_sustained_elapsed","lts__d_sectors.max.pct_of_peak_sustained_elapsed","lts__d_sectors.min.pct_of_peak_sustained_elapsed","lts__d_sectors.sum.pct_of_peak_sustained_elapsed","lts__gcomp_input_sectors.avg","lts__gcomp_input_sectors.max","lts__gcomp_input_sectors.min","lts__gcomp_input_sectors.sum","lts__lts2xbar_cycles_active.avg.pct_of_peak_sustained_elapsed","lts__lts2xbar_cycles_active.max.pct_of_peak_sustained_elapsed","lts__lts2xbar_cycles_active.min.pct_of_peak_sustained_elapsed","lts__lts2xbar_cycles_active.sum.pct_of_peak_sustained_elapsed","lts__t_sector_hit_rate.pct","lts__t_sectors.avg.pct_of_peak_sustained_elapsed","lts__t_sectors.max.pct_of_peak_sustained_elapsed","lts__t_sectors.min.pct_of_peak_sustained_elapsed","lts__t_sectors.sum.pct_of_peak_sustained_elapsed","lts__t_tag_requests.avg.pct_of_peak_sustained_elapsed","lts__t_tag_requests.max.pct_of_peak_sustained_elapsed","lts__t_tag_requests.min.pct_of_peak_sustained_elapsed","lts__t_tag_requests.sum.pct_of_peak_sustained_elapsed","lts__throughput.avg.pct_of_peak_sustained_elapsed","lts__throughput.max.pct_of_peak_sustained_elapsed","lts__throughput.min.pct_of_peak_sustained_elapsed","lts__throughput.sum.pct_of_peak_sustained_elapsed","lts__xbar2lts_cycles_active.avg.pct_of_peak_sustained_elapsed","lts__xbar2lts_cycles_active.max.pct_of_peak_sustained_elapsed","lts__xbar2lts_cycles_active.min.pct_of_peak_sustained_elapsed","lts__xbar2lts_cycles_active.sum.pct_of_peak_sustained_elapsed","numa__cpu_affinity","numa__dev_display_name_all","numa__id_cpu","numa__id_memory","nvlink__bandwidth","nvlink__count_logical","nvlink__count_physical","nvlink__destination_ports","nvlink__dev0Id","nvlink__dev0type","nvlink__dev1Id","nvlink__dev1type","nvlink__dev_display_name_all","nvlink__enabled_mask","nvlink__is_direct_link","nvlink__is_nvswitch_connected","nvlink__max_count","nvlink__peer_access","nvlink__peer_atomic","nvlink__source_ports","nvlink__system_access","nvlink__system_atomic","profiler__perfworks_session_reuse","profiler__replayer_bytes_mem_accessible.avg","profiler__replayer_bytes_mem_accessible.max","profiler__replayer_bytes_mem_accessible.min","profiler__replayer_bytes_mem_accessible.sum","profiler__replayer_bytes_mem_backed_up.avg","profiler__replayer_bytes_mem_backed_up.max","profiler__replayer_bytes_mem_backed_up.min","profiler__replayer_bytes_mem_backed_up.sum","profiler__replayer_passes","profiler__replayer_passes_type_warmup","sass__inst_executed_per_opcode","sass__inst_executed_per_opcode_category","sass__inst_executed_per_opcode_with_modifier_all","sass__inst_executed_per_opcode_with_modifier_selective","sass__inst_executed_register_spilling","sass__inst_executed_register_spilling_mem_local","sass__inst_executed_register_spilling_mem_shared","sass__inst_executed_register_spilling_op_read","sass__inst_executed_register_spilling_op_write","sass__thread_inst_executed_per_opcode_category","sass__thread_inst_executed_true_per_opcode","sass__thread_inst_executed_true_per_opcode_with_modifier_all","sass__thread_inst_executed_true_per_opcode_with_modifier_selective","sm__cycles_active.avg","sm__cycles_active.max","sm__cycles_active.min","sm__cycles_active.sum","sm__inst_executed.avg.pct_of_peak_sustained_elapsed","sm__inst_executed.max.pct_of_peak_sustained_elapsed","sm__inst_executed.min.pct_of_peak_sustained_elapsed","sm__inst_executed.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_adu.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_adu.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_adu.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_adu.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_cbu_pred_on_any.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_cbu_pred_on_any.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_cbu_pred_on_any.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_cbu_pred_on_any.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_ipa.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_ipa.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_ipa.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_ipa.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_lsu.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_lsu.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_lsu.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_lsu.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_tex.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_tex.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_tex.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_tex.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_uniform.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_uniform.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_uniform.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_uniform.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_xu.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_xu.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_xu.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_xu.sum.pct_of_peak_sustained_elapsed","sm__instruction_throughput_internal_activity.avg.pct_of_peak_sustained_elapsed","sm__instruction_throughput_internal_activity.max.pct_of_peak_sustained_elapsed","sm__instruction_throughput_internal_activity.min.pct_of_peak_sustained_elapsed","sm__instruction_throughput_internal_activity.sum.pct_of_peak_sustained_elapsed","sm__issue_active.avg.pct_of_peak_sustained_elapsed","sm__issue_active.max.pct_of_peak_sustained_elapsed","sm__issue_active.min.pct_of_peak_sustained_elapsed","sm__issue_active.sum.pct_of_peak_sustained_elapsed","sm__maximum_warps_avg_per_active_cycle","sm__maximum_warps_per_active_cycle_pct","sm__memory_throughput.avg.pct_of_peak_sustained_elapsed","sm__memory_throughput.max.pct_of_peak_sustained_elapsed","sm__memory_throughput.min.pct_of_peak_sustained_elapsed","sm__memory_throughput.sum.pct_of_peak_sustained_elapsed","sm__memory_throughput_internal_activity.avg.pct_of_peak_sustained_elapsed","sm__memory_throughput_internal_activity.max.pct_of_peak_sustained_elapsed","sm__memory_throughput_internal_activity.min.pct_of_peak_sustained_elapsed","sm__memory_throughput_internal_activity.sum.pct_of_peak_sustained_elapsed","sm__mio2rf_writeback_active.avg.pct_of_peak_sustained_elapsed","sm__mio2rf_writeback_active.max.pct_of_peak_sustained_elapsed","sm__mio2rf_writeback_active.min.pct_of_peak_sustained_elapsed","sm__mio2rf_writeback_active.sum.pct_of_peak_sustained_elapsed","sm__mio_inst_issued.avg.pct_of_peak_sustained_elapsed","sm__mio_inst_issued.max.pct_of_peak_sustained_elapsed","sm__mio_inst_issued.min.pct_of_peak_sustained_elapsed","sm__mio_inst_issued.sum.pct_of_peak_sustained_elapsed","sm__mio_pq_read_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__mio_pq_read_cycles_active.max.pct_of_peak_sustained_elapsed","sm__mio_pq_read_cycles_active.min.pct_of_peak_sustained_elapsed","sm__mio_pq_read_cycles_active.sum.pct_of_peak_sustained_elapsed","sm__mio_pq_write_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__mio_pq_write_cycles_active.max.pct_of_peak_sustained_elapsed","sm__mio_pq_write_cycles_active.min.pct_of_peak_sustained_elapsed","sm__mio_pq_write_cycles_active.sum.pct_of_peak_sustained_elapsed","sm__pipe_aluheavy_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__pipe_aluheavy_cycles_active.max.pct_of_peak_sustained_elapsed","sm__pipe_aluheavy_cycles_active.min.pct_of_peak_sustained_elapsed","sm__pipe_aluheavy_cycles_active.sum.pct_of_peak_sustained_elapsed","sm__pipe_fma_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__pipe_fma_cycles_active.max.pct_of_peak_sustained_elapsed","sm__pipe_fma_cycles_active.min.pct_of_peak_sustained_elapsed","sm__pipe_fma_cycles_active.sum.pct_of_peak_sustained_elapsed","sm__pipe_fmaheavy_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__pipe_fmaheavy_cycles_active.max.pct_of_peak_sustained_elapsed","sm__pipe_fmaheavy_cycles_active.min.pct_of_peak_sustained_elapsed","sm__pipe_fmaheavy_cycles_active.sum.pct_of_peak_sustained_elapsed","sm__pipe_fp64_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__pipe_fp64_cycles_active.max.pct_of_peak_sustained_elapsed","sm__pipe_fp64_cycles_active.min.pct_of_peak_sustained_elapsed","sm__pipe_fp64_cycles_active.sum.pct_of_peak_sustained_elapsed","sm__pipe_tensor_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__pipe_tensor_cycles_active.max.pct_of_peak_sustained_elapsed","sm__pipe_tensor_cycles_active.min.pct_of_peak_sustained_elapsed","sm__pipe_tensor_cycles_active.sum.pct_of_peak_sustained_elapsed","sm__pipe_tma_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__pipe_tma_cycles_active.max.pct_of_peak_sustained_elapsed","sm__pipe_tma_cycles_active.min.pct_of_peak_sustained_elapsed","sm__pipe_tma_cycles_active.sum.pct_of_peak_sustained_elapsed","sm__throughput.avg.pct_of_peak_sustained_elapsed","sm__throughput.max.pct_of_peak_sustained_elapsed","sm__throughput.min.pct_of_peak_sustained_elapsed","sm__throughput.sum.pct_of_peak_sustained_elapsed","sm__warps_active.avg.pct_of_peak_sustained_active","sm__warps_active.avg.per_cycle_active","sm__warps_active.max.pct_of_peak_sustained_active","sm__warps_active.max.per_cycle_active","sm__warps_active.min.pct_of_peak_sustained_active","sm__warps_active.min.per_cycle_active","sm__warps_active.sum.pct_of_peak_sustained_active","sm__warps_active.sum.per_cycle_active","smsp__average_warp_latency_per_inst_issued.ratio","smsp__average_warps_active_per_inst_executed.ratio","smsp__average_warps_issue_stalled_barrier_per_issue_active.ratio","smsp__average_warps_issue_stalled_branch_resolving_per_issue_active.ratio","smsp__average_warps_issue_stalled_dispatch_stall_per_issue_active.ratio","smsp__average_warps_issue_stalled_drain_per_issue_active.ratio","smsp__average_warps_issue_stalled_lg_throttle_per_issue_active.ratio","smsp__average_warps_issue_stalled_long_scoreboard_per_issue_active.ratio","smsp__average_warps_issue_stalled_math_pipe_throttle_per_issue_active.ratio","smsp__average_warps_issue_stalled_membar_per_issue_active.ratio","smsp__average_warps_issue_stalled_mio_throttle_per_issue_active.ratio","smsp__average_warps_issue_stalled_misc_per_issue_active.ratio","smsp__average_warps_issue_stalled_no_instruction_per_issue_active.ratio","smsp__average_warps_issue_stalled_not_selected_per_issue_active.ratio","smsp__average_warps_issue_stalled_selected_per_issue_active.ratio","smsp__average_warps_issue_stalled_short_scoreboard_per_issue_active.ratio","smsp__average_warps_issue_stalled_sleeping_per_issue_active.ratio","smsp__average_warps_issue_stalled_tex_throttle_per_issue_active.ratio","smsp__average_warps_issue_stalled_wait_per_issue_active.ratio","smsp__issue_active.avg.pct_of_peak_sustained_active","smsp__issue_active.avg.per_cycle_active","smsp__issue_active.max.pct_of_peak_sustained_active","smsp__issue_active.max.per_cycle_active","smsp__issue_active.min.pct_of_peak_sustained_active","smsp__issue_active.min.per_cycle_active","smsp__issue_active.sum.pct_of_peak_sustained_active","smsp__issue_active.sum.per_cycle_active","smsp__issue_inst0.avg.pct_of_peak_sustained_active","smsp__issue_inst0.max.pct_of_peak_sustained_active","smsp__issue_inst0.min.pct_of_peak_sustained_active","smsp__issue_inst0.sum.pct_of_peak_sustained_active","smsp__maximum_warps_avg_per_active_cycle","smsp__thread_inst_executed_per_inst_executed.ratio","smsp__thread_inst_executed_pred_on_per_inst_executed.ratio","smsp__warps_active.avg.peak_sustained","smsp__warps_active.avg.per_cycle_active","smsp__warps_active.max.peak_sustained","smsp__warps_active.max.per_cycle_active","smsp__warps_active.min.peak_sustained","smsp__warps_active.min.per_cycle_active","smsp__warps_active.sum.peak_sustained","smsp__warps_active.sum.per_cycle_active","smsp__warps_eligible.avg.per_cycle_active","smsp__warps_eligible.max.per_cycle_active","smsp__warps_eligible.min.per_cycle_active","smsp__warps_eligible.sum.per_cycle_active" +"","","","","","","","","","","","","","","%","","%","%/register","%/byte","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","cycle","hz","cycle","hz","cycle","hz","cycle","hz","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","ns","ns","ns","ns","","","","","block","block","block","block","","","","","%","%","%","%","inst","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","","block","block","block","","","","","cluster","block","","","","","","","","","","","","%","%","block","block","block","block","block","","","","","","byte","","","","","register/thread","register/thread","byte","byte/block","byte/block","byte/block","byte/block","byte/block","SM","","","thread","","","","","","","","","%","","%","%","%","%","%","%","%","%","%","%","%","%","sector","sector","sector","sector","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","","","","","","","","","","","","","","","","","","","","","","","","byte","byte","byte","byte","byte","byte","byte","byte","pass","pass","","","","","inst","byte","byte","inst","inst","","","","","cycle","cycle","cycle","cycle","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","warp","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","warp","%","warp","%","warp","%","warp","cycle","cycle","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","%","","%","","%","","%","","%","%","%","%","warp","","","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp" +"0","60","python3.12","127.0.0.1","cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu","1","7","(384, 1, 1)","(296, 168, 1)","0","12.1","0","1","0","no data","625","316","4225","4975","30000","432","1","0","1","1","1","0","1","0","1","0","443","2418000","1","12","1","0","1","1","1","No-CC","1","1","1","0","0","NVIDIA GB10","0","0","256","4","1","1","256","1","0","0","100","1","772935902","11794","32000","16","0","4","2500","16","161","4318","0","1","0","0","1","0","1","443","1","1","1","25165824","24","24","24","1","1","134217728","1024","1024","64","24","2418000","2147483647","65535","65535","4","1","8533000","18874368","2147483647","65536","65536","255","49152","101376","102400","1024","1536","48","12","2048","32768","32768","65536","32768","2048","32768","131072","16384","16384","16384","2046","32768","32768","2048","32768","268435456","32768","131072","32768","32768","65536","32768","2048","32768","65000","2097120","131072","32768","32768","131072","16384","32768","16384","8192","16384","8192","2046","32768","32768","4","8533000","1","1","0","0","0","0","48","6","4","1","0","1","1","1","0","15","1","0","1024","12","64","1","1","512","0","1","512","32","65536","130661769216","1","1","1","32","56409993.25","2138683834.25","56422407","2139154479.34","56373386","2137295935.95","225639973","8554735336.99","77.10","77.13","77.06","77.10","38.73","38.77","38.68","38.73","73.98","74.18","73.77","73.98","0","0","0","0","77.10","77.13","77.06","77.10","26376032","26376032","26376032","26376032","49680","49680","49680","49680","49680","49680","49680","49680","49728","49728","49728","49728","0.40","0.41","0.40","0.40","1920404495","16.90","17.06","16.75","16.90","5.79","5.85","5.74","5.79","41.27","41.67","40.91","41.27","0","0","0","0","0","0","0","0","37.40","37.65","37.18","37.40","26.41","26.62","26.21","26.41","9.00","9.06","8.94","9.00","44.43","44.82","44.13","44.43","90","0","0","0","0","0.11","0.11","0.11","0.11","43.23","43.60","42.93","43.23","8","384","1","1","384","0","0","0","0","8","PolicySpread","0","1","0","CachePreferNone","1","296","168","1","49728","","0","0","3","24","1","1","4","300","157","0","2028","2388","4718592","0","0","0","0","168","168","102400","89088","89088","1024","88064","0","48","1024","7","19095552","24","all","0","0","0","0","0","1036","0","0","0","0","0","0","0","0","0","0","42.74","42.76","42.72","42.74","0","0","0","0","73.98","74.18","73.77","73.98","89.11","77.10","77.13","77.06","77.10","38.55","38.57","38.53","38.55","77.10","77.13","77.06","77.10","34.08","34.12","34.05","34.08","1","1","1","1","0","0","0","0","0","0","0","0","1","0","0","0","0","0","0","0","0","0","0","7414590282","7414590282","7414590282","126048034794","7414590282","7414590282","7414590282","126048034794","17","0","1920404495","1920404495","1920404495","1920404495","0","no data","no data","0","0","0","0","0","0","57979728.96","58027199","57917994","2783026990","17.92","18.05","17.79","17.92","4.56","4.58","4.55","4.56","0.66","0.66","0.65","0.66","0","0","0","0","26.41","26.64","26.18","26.41","0","0","0","0","3.20","3.22","3.18","3.20","0","0","0","0","0","0","0","0","18.03","18.19","17.90","18.03","12","25","26.41","26.64","26.18","26.41","0.12","0.12","0.12","0.12","9.40","9.49","9.33","9.40","10.45","10.53","10.37","10.45","3.86","3.89","3.82","3.86","3.86","3.89","3.82","3.86","6.91","6.98","6.85","6.91","0.85","0.86","0.85","0.85","2.77","2.80","2.75","2.77","0","0","0","0","78.99","79.60","78.38","78.99","0.55","0.56","0.55","0.55","78.99","79.60","78.38","78.99","20.83","10.00","20.85","10.01","20.81","9.99","20.83","480.00","13.67","13.75","0.23","0.16","0.04","0.00","0","0.61","3.82","0","0.07","0.02","0.08","0.31","1.00","0.08","3.93","0","3.78","18.29","0.18","20.05","0.20","17.48","0.17","18.29","35.12","81.71","80.01","82.42","81.71","3","31.24","25.64","12","2.50","12","3.00","12","2.00","2304","480.00","0.24","0.26","0.23","45.95" +"1","60","python3.12","127.0.0.1","void qk_int_sv_f8_attn_kernel<128, 64, 32, 64, 128, 1, 2, 2, float, 1, __nv_bfloat16, 1, 0, 0, 1, 0, 1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)","1","7","(32, 4, 1)","(296, 56, 1)","0","12.1","0","1","1458688","no data","304","202","5633","2400","15200","432","1","0","1","1","1","0","1","0","1","0","443","2418000","1","12","1","0","1","1","1","No-CC","1","1","1","0","0","NVIDIA GB10","0","0","256","4","1","1","256","1","0","0","100","1","772935902","11794","32000","16","0","4","2500","16","161","4318","0","1","0","0","1","0","1","443","1","1","1","25165824","24","24","24","1","1","134217728","1024","1024","64","24","2418000","2147483647","65535","65535","4","1","8533000","18874368","2147483647","65536","65536","255","49152","101376","102400","1024","1536","48","12","2048","32768","32768","65536","32768","2048","32768","131072","16384","16384","16384","2046","32768","32768","2048","32768","268435456","32768","131072","32768","32768","65536","32768","2048","32768","65000","2097120","131072","32768","32768","131072","16384","32768","16384","8192","16384","8192","2046","32768","32768","4","8533000","1","1","0","0","0","0","48","6","4","1","0","1","1","1","0","15","1","0","1024","12","64","1","1","512","0","1","512","32","65536","130661769216","1","1","1","32","553521770","2137182697.80","553599951","2137484559.61","553288532","2136282151.08","2214087080","8548730791.21","31.59","31.60","31.58","31.59","23.67","23.67","23.66","23.67","31.46","31.47","31.46","31.46","0","0","0","0","31.59","31.60","31.58","31.59","258996000","258996000","258996000","258996000","0","0","0","0","0","0","0","0","0","0","0","0","0.00","0.00","0.00","0.00","38322384408","11.82","11.84","11.78","11.82","2.38","2.39","2.37","2.38","26.66","26.71","26.55","26.66","0","0","0","0","0","0","0","0","25.10","25.15","25.01","25.10","19.55","19.59","19.47","19.55","7.17","7.18","7.14","7.17","18.92","18.95","18.84","18.92","0.80","0","0","0","0","0.00","0.00","0.00","0.00","26.89","26.94","26.78","26.89","1","32","4","1","128","0","0","0","0","8","PolicySpread","0","1","0","CachePreferNone","1","296","56","1","16576","","0","0","24","24","2","3","12","152","101","0","2732","1200","4718592","0","0","0","0","255","256","102400","33792","33792","1024","32768","0","48","1024","7","2121728","24","all","0","0","0","0","0","172.67","0","0","0","0","0","0","0","0","0","0","15.98","15.99","15.98","15.98","0","0","0","0","31.46","31.47","31.46","31.46","98.84","31.59","31.60","31.58","31.59","23.65","23.66","23.65","23.65","31.59","31.60","31.58","31.59","23.84","23.85","23.83","23.84","1","1","1","1","0","0","0","0","0","0","0","0","1","0","0","0","0","0","0","0","0","0","0","7414590282","7414590282","7414590282","126048034794","7414590282","7414590282","7414590282","126048034794","17","0","38322384408","38322384408","38322384408","38322384408","1458688","no data","no data","0","0","0","0","0","0","548783193.98","549710880","546269918","26341593311","36.06","36.13","35.92","36.06","1.18","1.19","1.18","1.18","0.00","0.00","0.00","0.00","0","0","0","0","19.55","19.59","19.47","19.55","0","0","0","0","0.22","0.22","0.22","0.22","20.06","20.10","19.98","20.06","0","0","0","0","36.06","36.13","35.92","36.06","8","16.67","19.55","19.59","19.47","19.55","0.00","0.00","0.00","0.00","6.28","6.29","6.25","6.28","6.91","6.92","6.88","6.91","1.22","1.22","1.21","1.22","1.22","1.22","1.21","1.22","12.94","12.97","12.89","12.94","15.00","15.03","14.94","15.00","19.23","19.27","19.16","19.23","0","0","0","0","75.51","75.66","75.22","75.51","0","0","0","0","75.51","75.66","75.22","75.51","16.65","7.99","16.69","8.01","16.59","7.96","16.65","383.64","5.48","5.48","0.19","0.00","0.16","0.00","0.00","0.14","1.24","0","0.12","0.08","0.03","0.26","1","0.25","0","0","2.01","36.49","0.36","36.57","0.37","36.35","0.36","36.49","70.07","63.51","63.63","63.21","63.51","2","32","31.80","12","2.00","12","2.00","12","1.99","2304","383.82","0.46","0.46","0.46","88.19" +"2","60","python3.12","127.0.0.1","cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu","1","7","(384, 1, 1)","(296, 42, 1)","0","12.1","0","1","0","no data","625","316","4225","4975","30000","432","1","0","1","1","1","0","1","0","1","0","443","2418000","1","12","1","0","1","1","1","No-CC","1","1","1","0","0","NVIDIA GB10","0","0","256","4","1","1","256","1","0","0","100","1","772935902","11794","32000","16","0","4","2500","16","161","4318","0","1","0","0","1","0","1","443","1","1","1","25165824","24","24","24","1","1","134217728","1024","1024","64","24","2418000","2147483647","65535","65535","4","1","8533000","18874368","2147483647","65536","65536","255","49152","101376","102400","1024","1536","48","12","2048","32768","32768","65536","32768","2048","32768","131072","16384","16384","16384","2046","32768","32768","2048","32768","268435456","32768","131072","32768","32768","65536","32768","2048","32768","65000","2097120","131072","32768","32768","131072","16384","32768","16384","8192","16384","8192","2046","32768","32768","4","8533000","1","1","0","0","0","0","48","6","4","1","0","1","1","1","0","15","1","0","1024","12","64","1","1","512","0","1","512","32","65536","130661769216","1","1","1","32","18139065.50","2138750195.73","18142556","2139161755.38","18128846","2137545229.70","72556262","8555000782.91","79.10","79.16","79.05","79.10","39.71","39.78","39.65","39.71","76.67","76.90","76.49","76.67","0","0","0","0","79.10","79.16","79.05","79.10","8481152","8481152","8481152","8481152","12384","12384","12384","12384","12384","12384","12384","12384","12432","12432","12432","12432","0.34","0.34","0.33","0.34","622830747","17.46","17.66","17.32","17.46","5.94","6.01","5.89","5.94","42.41","42.90","42.08","42.41","0","0","0","0","0","0","0","0","38.75","39.05","38.45","38.75","27.10","27.30","26.89","27.10","11.08","11.16","10.99","11.08","46.06","46.41","45.70","46.06","90","0","0","0","0","0.08","0.09","0.08","0.08","42.67","43.00","42.34","42.67","8","384","1","1","384","0","0","0","0","8","PolicySpread","0","1","0","CachePreferNone","1","296","42","1","12432","","0","0","3","24","1","1","4","300","157","0","2028","2388","4718592","0","0","0","0","168","168","102400","89088","89088","1024","88064","0","48","1024","7","4773888","24","all","0","0","0","0","0","259","0","0","0","0","0","0","0","0","0","0","43.53","43.57","43.50","43.53","0","0","0","0","76.67","76.90","76.49","76.67","89.88","79.10","79.16","79.05","79.10","39.55","39.59","39.53","39.55","79.10","79.16","79.05","79.10","41.21","41.27","41.19","41.21","1","1","1","1","0","0","0","0","0","0","0","0","1","0","0","0","0","0","0","0","0","0","0","7414590282","7414590282","7414590282","126048034794","7414590282","7414590282","7414590282","126048034794","17","0","622830747","622830747","622830747","622830747","0","no data","no data","0","0","0","0","0","0","19579361.17","19615758","19538215","939809336","18.09","18.22","17.95","18.09","4.55","4.57","4.52","4.55","0.61","0.62","0.60","0.61","0","0","0","0","27.10","27.30","26.89","27.10","0","0","0","0","3.34","3.37","3.32","3.34","0","0","0","0","0","0","0","0","18.19","18.39","18.06","18.19","12","25","27.10","27.30","26.89","27.10","0.09","0.09","0.09","0.09","9.73","9.84","9.66","9.73","10.67","10.75","10.59","10.67","3.88","3.92","3.84","3.88","3.88","3.92","3.84","3.88","6.79","6.87","6.74","6.79","0.80","0.81","0.79","0.80","2.52","2.54","2.50","2.52","0","0","0","0","81.88","82.51","81.25","81.88","0.55","0.55","0.54","0.55","81.88","82.51","81.25","81.88","20.83","10.00","20.87","10.02","20.79","9.98","20.83","480.00","14.28","14.36","0.22","0.16","0.04","0.00","0","0.60","3.90","0","0.06","0.02","0.07","0.31","1.00","0.07","4.08","0","3.85","17.51","0.18","19.18","0.19","16.76","0.17","17.51","33.62","82.49","81.06","83.06","82.49","3","31.23","25.41","12","2.50","12","3.01","12","2.00","2304","480.00","0.23","0.25","0.22","43.96" +"3","60","python3.12","127.0.0.1","cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu","1","7","(384, 1, 1)","(296, 224, 1)","0","12.1","0","1","0","no data","625","316","4225","4975","30000","432","1","0","1","1","1","0","1","0","1","0","443","2418000","1","12","1","0","1","1","1","No-CC","1","1","1","0","0","NVIDIA GB10","0","0","256","4","1","1","256","1","0","0","100","1","772935902","11794","32000","16","0","4","2500","16","161","4318","0","1","0","0","1","0","1","443","1","1","1","25165824","24","24","24","1","1","134217728","1024","1024","64","24","2418000","2147483647","65535","65535","4","1","8533000","18874368","2147483647","65536","65536","255","49152","101376","102400","1024","1536","48","12","2048","32768","32768","65536","32768","2048","32768","131072","16384","16384","16384","2046","32768","32768","2048","32768","268435456","32768","131072","32768","32768","65536","32768","2048","32768","65000","2097120","131072","32768","32768","131072","16384","32768","16384","8192","16384","8192","2046","32768","32768","4","8533000","1","1","0","0","0","0","48","6","4","1","0","1","1","1","0","15","1","0","1024","12","64","1","1","512","0","1","512","32","65536","130661769216","1","1","1","32","77303952","2137085387.93","77322941","2137610343.16","77247220","2135517018.84","309215808","8548341551.71","74.99","75.02","74.96","74.99","37.66","37.69","37.61","37.66","71.95","72.13","71.77","71.95","0","0","0","0","74.99","75.02","74.96","74.99","36172608","36172608","36172608","36172608","66256","66256","66256","66256","66256","66256","66256","66256","66304","66304","66304","66304","0.39","0.40","0.39","0.39","2560592083","16.44","16.56","16.34","16.44","5.63","5.67","5.60","5.63","40.15","40.43","39.92","40.15","0","0","0","0","0","0","0","0","36.39","36.72","36.09","36.39","25.70","25.92","25.51","25.70","8.75","8.83","8.69","8.75","43.23","43.59","42.87","43.23","90","0","0","0","0","0.11","0.11","0.10","0.11","43.30","43.66","42.94","43.30","8","384","1","1","384","0","0","0","0","8","PolicySpread","0","1","0","CachePreferNone","1","296","224","1","66304","","0","0","3","24","1","1","4","300","157","0","2028","2388","4718592","0","0","0","0","168","168","102400","89088","89088","1024","88064","0","48","1024","7","25460736","24","all","0","0","0","0","0","1381.33","0","0","0","0","0","0","0","0","0","0","41.57","41.59","41.55","41.57","0","0","0","0","71.95","72.13","71.77","71.95","89.11","74.99","75.02","74.96","74.99","37.50","37.52","37.48","37.50","74.99","75.02","74.96","74.99","33.14","33.17","33.12","33.14","1","1","1","1","0","0","0","0","0","0","0","0","1","0","0","0","0","0","0","0","0","0","0","7414590282","7414590282","7414590282","126048034794","7414590282","7414590282","7414590282","126048034794","17","0","2560592083","2560592083","2560592083","2560592083","0","no data","no data","0","0","0","0","0","0","77183115.25","77224260","77101423","3704789532","17.44","17.58","17.31","17.44","4.44","4.46","4.43","4.44","0.64","0.64","0.63","0.64","0","0","0","0","25.70","25.90","25.45","25.70","0","0","0","0","3.12","3.14","3.09","3.12","0","0","0","0","0","0","0","0","17.54","17.66","17.39","17.54","12","25","25.70","25.90","25.45","25.70","0.12","0.12","0.12","0.12","9.15","9.22","9.06","9.15","10.17","10.24","10.08","10.17","3.76","3.78","3.73","3.76","3.76","3.78","3.73","3.76","6.72","6.77","6.68","6.72","0.83","0.84","0.83","0.83","2.69","2.71","2.67","2.69","0","0","0","0","76.85","77.39","76.22","76.85","0.54","0.54","0.53","0.54","76.85","77.39","76.22","76.85","20.83","10.00","20.84","10.01","20.81","9.99","20.83","480.00","14.38","14.46","0.24","0.16","0.04","0.00","0","0.61","3.82","0","0.07","0.02","0.08","0.31","1.00","0.08","4.41","0","3.78","17.39","0.17","19.03","0.19","16.60","0.17","17.39","33.38","82.61","81.02","83.29","82.61","3","31.25","25.64","12","2.50","12","3.00","12","2.00","2304","480.00","0.23","0.25","0.22","43.68" +"4","60","python3.12","127.0.0.1","cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu","1","7","(384, 1, 1)","(296, 42, 1)","0","12.1","0","1","0","no data","625","316","4225","4975","30000","432","1","0","1","1","1","0","1","0","1","0","443","2418000","1","12","1","0","1","1","1","No-CC","1","1","1","0","0","NVIDIA GB10","0","0","256","4","1","1","256","1","0","0","100","1","772935902","11794","32000","16","0","4","2500","16","161","4318","0","1","0","0","1","0","1","443","1","1","1","25165824","24","24","24","1","1","134217728","1024","1024","64","24","2418000","2147483647","65535","65535","4","1","8533000","18874368","2147483647","65536","65536","255","49152","101376","102400","1024","1536","48","12","2048","32768","32768","65536","32768","2048","32768","131072","16384","16384","16384","2046","32768","32768","2048","32768","268435456","32768","131072","32768","32768","65536","32768","2048","32768","65000","2097120","131072","32768","32768","131072","16384","32768","16384","8192","16384","8192","2046","32768","32768","4","8533000","1","1","0","0","0","0","48","6","4","1","0","1","1","1","0","15","1","0","1024","12","64","1","1","512","0","1","512","32","65536","130661769216","1","1","1","32","34588907.25","2135252700.54","34597216","2135765618.77","34564347","2133736539.89","138355629","8541010802.16","81.58","81.61","81.56","81.58","40.95","41.00","40.90","40.95","80.30","80.45","80.15","80.30","0","0","0","0","81.58","81.61","81.56","81.58","16198976","16198976","16198976","16198976","12384","12384","12384","12384","12384","12384","12384","12384","12432","12432","12432","12432","0.22","0.22","0.22","0.22","1192937654","18.21","18.35","18.07","18.21","6.13","6.18","6.09","6.13","43.97","44.31","43.63","43.97","0","0","0","0","0","0","0","0","40.62","40.93","40.30","40.62","27.96","28.29","27.64","27.96","12.18","12.28","12.07","12.18","48.31","48.68","47.93","48.31","90","0","0","0","0","0.04","0.04","0.04","0.04","46.09","46.45","45.74","46.09","8","384","1","1","384","0","0","0","0","8","PolicySpread","0","1","0","CachePreferNone","1","296","42","1","12432","","0","0","3","24","1","1","4","300","157","0","2028","2388","4718592","0","0","0","0","168","168","102400","89088","89088","1024","88064","0","48","1024","7","4773888","24","all","0","0","0","0","0","259","0","0","0","0","0","0","0","0","0","0","44.41","44.42","44.40","44.41","0","0","0","0","80.30","80.45","80.15","80.30","91.07","81.58","81.61","81.56","81.58","40.80","40.82","40.79","40.80","81.58","81.61","81.56","81.58","40.58","40.62","40.56","40.58","1","1","1","1","0","0","0","0","0","0","0","0","1","0","0","0","0","0","0","0","0","0","0","7414590282","7414590282","7414590282","126048034794","7414590282","7414590282","7414590282","126048034794","17","0","1192937654","1192937654","1192937654","1192937654","0","no data","no data","0","0","0","0","0","0","36250542.50","36317254","36167629","1740026040","18.18","18.38","17.98","18.18","4.45","4.47","4.42","4.45","0.55","0.55","0.54","0.55","0","0","0","0","27.96","28.18","27.75","27.96","0","0","0","0","3.54","3.57","3.51","3.54","0","0","0","0","0","0","0","0","18.29","18.49","18.16","18.29","12","25","27.96","28.18","27.75","27.96","0.05","0.05","0.05","0.05","10.18","10.30","10.10","10.18","10.92","11.01","10.84","10.92","3.90","3.95","3.86","3.90","3.90","3.95","3.86","3.90","6.53","6.58","6.48","6.53","0.70","0.71","0.70","0.70","2.10","2.12","2.08","2.10","0","0","0","0","85.88","86.54","85.21","85.88","0.54","0.55","0.53","0.54","85.88","86.54","85.21","85.88","20.83","10.00","20.87","10.02","20.79","9.98","20.83","480.00","14.34","14.43","0.17","0.16","0.04","0.00","0","0.60","4.06","0","0.05","0.01","0.07","0.31","1.00","0.05","3.81","0","3.98","17.44","0.17","19.04","0.19","16.72","0.17","17.44","33.48","82.56","81.15","83.05","82.56","3","31.22","25.05","12","2.50","12","3.01","12","2.00","2304","480.00","0.23","0.25","0.22","43.74" diff --git a/benchmarks/gb10-post-fc2-block24-targeted-20260826.ncu-rep b/benchmarks/gb10-post-fc2-block24-targeted-20260826.ncu-rep new file mode 100644 index 0000000..8e37b0f Binary files /dev/null and b/benchmarks/gb10-post-fc2-block24-targeted-20260826.ncu-rep differ diff --git a/benchmarks/gb10-post-fc2-block24-targeted-capture-20260826.json b/benchmarks/gb10-post-fc2-block24-targeted-capture-20260826.json new file mode 100644 index 0000000..995fd1b --- /dev/null +++ b/benchmarks/gb10-post-fc2-block24-targeted-capture-20260826.json @@ -0,0 +1,31 @@ +{ + "block_index": 24, + "hidden_shape": [ + 37810, + 5376 + ], + "segments": [ + [ + 0, + 100, + 1 + ], + [ + 100, + 514, + 2 + ], + [ + 514, + 37810, + 0 + ] + ], + "fused_elementwise": true, + "fused_nvfp4_modulation": true, + "fused_nvfp4_swiglu": true, + "nvfp4_scale_backend": "vortex", + "sage_qkv_layout": "strided_nhd", + "module_forward_checksum": 303055616.0, + "capture": "one warmed block between cudaProfilerStart/Stop" +} \ No newline at end of file diff --git a/benchmarks/gb10-post-fc2-block24-targeted-summary-20260826.json b/benchmarks/gb10-post-fc2-block24-targeted-summary-20260826.json new file mode 100644 index 0000000..83f0038 --- /dev/null +++ b/benchmarks/gb10-post-fc2-block24-targeted-summary-20260826.json @@ -0,0 +1,219 @@ +{ + "source_csv": "benchmarks\\gb10-post-fc2-block24-targeted-20260826.csv", + "source_traffic_csv": "benchmarks\\gb10-post-fc2-block24-targeted-traffic-20260826.csv", + "launch_order_contract": [ + "qkv", + "sage2", + "attention_output", + "fc1", + "fc2" + ], + "cache_control": "none (warmed/uncontrolled cache, as reported by NCU)", + "metrics": { + "qkv": { + "launch_id": 0, + "kernel_name": "cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu", + "grid_size": "(296, 168, 1)", + "block_size": "(384, 1, 1)", + "duration_ns": 26376032.0, + "registers_per_thread": 168.0, + "achieved_occupancy_percent": 20.83, + "eligible_warps_per_scheduler": 0.24, + "issue_active_percent": 18.29, + "sm_throughput_percent": 78.99, + "tensor_pipe_active_percent": 78.99, + "l2_requested_bytes": 40128950176.0, + "l2_hit_rate_percent": 89.11, + "l2_throughput_percent": 77.1, + "memory_throughput_percent": 76.92, + "local_spilling_requests": 0.0, + "top_scheduler_stalls": [ + { + "reason": "sleeping", + "warps_per_issue_active": 3.93 + }, + { + "reason": "math_pipe_throttle", + "warps_per_issue_active": 3.82 + }, + { + "reason": "wait", + "warps_per_issue_active": 3.78 + }, + { + "reason": "selected", + "warps_per_issue_active": 1.0 + }, + { + "reason": "long_scoreboard", + "warps_per_issue_active": 0.61 + } + ] + }, + "sage2": { + "launch_id": 1, + "kernel_name": "void qk_int_sv_f8_attn_kernel<128, 64, 32, 64, 128, 1, 2, 2, float, 1, __nv_bfloat16, 1, 0, 0, 1, 0, 1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)", + "grid_size": "(296, 56, 1)", + "block_size": "(32, 4, 1)", + "duration_ns": 258996000.0, + "registers_per_thread": 255.0, + "achieved_occupancy_percent": 16.65, + "eligible_warps_per_scheduler": 0.46, + "issue_active_percent": 36.49, + "sm_throughput_percent": 75.51, + "tensor_pipe_active_percent": 75.51, + "l2_requested_bytes": 161497982464.0, + "l2_hit_rate_percent": 98.85, + "l2_throughput_percent": 31.59, + "memory_throughput_percent": 31.74, + "local_spilling_requests": 1458688.0, + "top_scheduler_stalls": [ + { + "reason": "wait", + "warps_per_issue_active": 2.01 + }, + { + "reason": "math_pipe_throttle", + "warps_per_issue_active": 1.24 + }, + { + "reason": "selected", + "warps_per_issue_active": 1.0 + }, + { + "reason": "not_selected", + "warps_per_issue_active": 0.26 + }, + { + "reason": "short_scoreboard", + "warps_per_issue_active": 0.25 + } + ] + }, + "attention_output": { + "launch_id": 2, + "kernel_name": "cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu", + "grid_size": "(296, 42, 1)", + "block_size": "(384, 1, 1)", + "duration_ns": 8481152.0, + "registers_per_thread": 168.0, + "achieved_occupancy_percent": 20.83, + "eligible_warps_per_scheduler": 0.23, + "issue_active_percent": 17.51, + "sm_throughput_percent": 81.88, + "tensor_pipe_active_percent": 81.88, + "l2_requested_bytes": 13241354176.0, + "l2_hit_rate_percent": 89.89, + "l2_throughput_percent": 79.1, + "memory_throughput_percent": 79.51, + "local_spilling_requests": 0.0, + "top_scheduler_stalls": [ + { + "reason": "sleeping", + "warps_per_issue_active": 4.08 + }, + { + "reason": "math_pipe_throttle", + "warps_per_issue_active": 3.9 + }, + { + "reason": "wait", + "warps_per_issue_active": 3.85 + }, + { + "reason": "selected", + "warps_per_issue_active": 1.0 + }, + { + "reason": "long_scoreboard", + "warps_per_issue_active": 0.6 + } + ] + }, + "fc1": { + "launch_id": 3, + "kernel_name": "cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu", + "grid_size": "(296, 224, 1)", + "block_size": "(384, 1, 1)", + "duration_ns": 36172608.0, + "registers_per_thread": 168.0, + "achieved_occupancy_percent": 20.83, + "eligible_warps_per_scheduler": 0.23, + "issue_active_percent": 17.39, + "sm_throughput_percent": 76.85, + "tensor_pipe_active_percent": 76.85, + "l2_requested_bytes": 53505320928.0, + "l2_hit_rate_percent": 89.11, + "l2_throughput_percent": 74.99, + "memory_throughput_percent": 74.85, + "local_spilling_requests": 0.0, + "top_scheduler_stalls": [ + { + "reason": "sleeping", + "warps_per_issue_active": 4.41 + }, + { + "reason": "math_pipe_throttle", + "warps_per_issue_active": 3.82 + }, + { + "reason": "wait", + "warps_per_issue_active": 3.78 + }, + { + "reason": "selected", + "warps_per_issue_active": 1.0 + }, + { + "reason": "long_scoreboard", + "warps_per_issue_active": 0.61 + } + ] + }, + "fc2": { + "launch_id": 4, + "kernel_name": "cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu", + "grid_size": "(296, 42, 1)", + "block_size": "(384, 1, 1)", + "duration_ns": 16198976.0, + "registers_per_thread": 168.0, + "achieved_occupancy_percent": 20.83, + "eligible_warps_per_scheduler": 0.23, + "issue_active_percent": 17.44, + "sm_throughput_percent": 85.88, + "tensor_pipe_active_percent": 85.88, + "l2_requested_bytes": 26072923840.0, + "l2_hit_rate_percent": 91.06, + "l2_throughput_percent": 81.58, + "memory_throughput_percent": 77.65, + "local_spilling_requests": 0.0, + "top_scheduler_stalls": [ + { + "reason": "math_pipe_throttle", + "warps_per_issue_active": 4.06 + }, + { + "reason": "wait", + "warps_per_issue_active": 3.98 + }, + { + "reason": "sleeping", + "warps_per_issue_active": 3.81 + }, + { + "reason": "selected", + "warps_per_issue_active": 1.0 + }, + { + "reason": "long_scoreboard", + "warps_per_issue_active": 0.6 + } + ] + } + }, + "units": { + "duration_ns": "ns", + "l2_requested_bytes": "lts__t_bytes.sum", + "throughput_and_hit_rate": "%" + } +} diff --git a/benchmarks/gb10-post-fc2-block24-targeted-traffic-20260826.csv b/benchmarks/gb10-post-fc2-block24-targeted-traffic-20260826.csv new file mode 100644 index 0000000..6fd15d5 --- /dev/null +++ b/benchmarks/gb10-post-fc2-block24-targeted-traffic-20260826.csv @@ -0,0 +1,7 @@ +"ID","Process ID","Process Name","Host Name","Kernel Name","Context","Stream","Block Size","Grid Size","Device","CC","c2clink__enabled_mask","c2clink__present","device__attribute_architecture","device__attribute_async_engine_count","device__attribute_can_flush_remote_writes","device__attribute_can_map_host_memory","device__attribute_can_tex2d_gather","device__attribute_can_use_64_bit_stream_mem_ops","device__attribute_can_use_64_bit_stream_mem_ops_v1","device__attribute_can_use_host_pointer_for_registered_mem","device__attribute_can_use_stream_mem_ops_v1","device__attribute_can_use_stream_wait_value_nor","device__attribute_can_use_stream_wait_value_nor_v1","device__attribute_chip","device__attribute_clock_rate","device__attribute_cluster_launch","device__attribute_compute_capability_major","device__attribute_compute_capability_minor","device__attribute_compute_mode","device__attribute_compute_preemption_supported","device__attribute_concurrent_kernels","device__attribute_concurrent_managed_access","device__attribute_confidential_computing_mode","device__attribute_cooperative_launch","device__attribute_cooperative_multi_device_launch","device__attribute_deferred_mapping_cuda_array_supported","device__attribute_device_index","device__attribute_direct_managed_mem_access_from_host","device__attribute_display_name","device__attribute_dma_buf_supported","device__attribute_ecc_enabled","device__attribute_fb_bus_width","device__attribute_fbp_count","device__attribute_generic_compression_supported","device__attribute_global_l1_cache_supported","device__attribute_global_memory_bus_width","device__attribute_gpu_direct_rdma_flush_writes_options","device__attribute_gpu_direct_rdma_supported","device__attribute_gpu_direct_rdma_with_cuda_vmm_supported","device__attribute_gpu_direct_rdma_writes_ordering","device__attribute_gpu_overlap","device__attribute_gpu_pci_device_id","device__attribute_gpu_pci_ext_device_id","device__attribute_gpu_pci_ext_downstream_link_rate","device__attribute_gpu_pci_ext_downstream_link_width","device__attribute_gpu_pci_ext_gen","device__attribute_gpu_pci_ext_gpu_gen","device__attribute_gpu_pci_ext_gpu_link_rate","device__attribute_gpu_pci_ext_gpu_link_width","device__attribute_gpu_pci_revision_id","device__attribute_gpu_pci_sub_system_id","device__attribute_handle_type_fabric_supported","device__attribute_handle_type_posix_file_descriptor_supported","device__attribute_handle_type_win32_handle_supported","device__attribute_handle_type_win32_kmt_handle_supported","device__attribute_host_native_atomic_supported","device__attribute_host_numa_id","device__attribute_host_register_supported","device__attribute_implementation","device__attribute_integrated","device__attribute_ipc_event_supported","device__attribute_kernel_exec_timeout","device__attribute_l2_cache_size","device__attribute_l2s_count","device__attribute_limits_max_cta_per_sm","device__attribute_limits_num_tpcs","device__attribute_local_l1_cache_supported","device__attribute_managed_memory","device__attribute_max_access_policy_window_size","device__attribute_max_block_dim_x","device__attribute_max_block_dim_y","device__attribute_max_block_dim_z","device__attribute_max_blocks_per_multiprocessor","device__attribute_max_gpu_frequency_khz","device__attribute_max_grid_dim_x","device__attribute_max_grid_dim_y","device__attribute_max_grid_dim_z","device__attribute_max_ipc_per_multiprocessor","device__attribute_max_ipc_per_scheduler","device__attribute_max_mem_frequency_khz","device__attribute_max_persisting_l2_cache_size","device__attribute_max_pitch","device__attribute_max_registers_per_block","device__attribute_max_registers_per_multiprocessor","device__attribute_max_registers_per_thread","device__attribute_max_shared_memory_per_block","device__attribute_max_shared_memory_per_block_optin","device__attribute_max_shared_memory_per_multiprocessor","device__attribute_max_threads_per_block","device__attribute_max_threads_per_multiprocessor","device__attribute_max_warps_per_multiprocessor","device__attribute_max_warps_per_scheduler","device__attribute_maximum_surface1d_layered_layers","device__attribute_maximum_surface1d_layered_width","device__attribute_maximum_surface1d_width","device__attribute_maximum_surface2d_height","device__attribute_maximum_surface2d_layered_height","device__attribute_maximum_surface2d_layered_layers","device__attribute_maximum_surface2d_layered_width","device__attribute_maximum_surface2d_width","device__attribute_maximum_surface3d_depth","device__attribute_maximum_surface3d_height","device__attribute_maximum_surface3d_width","device__attribute_maximum_surfacecubemap_layered_layers","device__attribute_maximum_surfacecubemap_layered_width","device__attribute_maximum_surfacecubemap_width","device__attribute_maximum_texture1d_layered_layers","device__attribute_maximum_texture1d_layered_width","device__attribute_maximum_texture1d_linear_width","device__attribute_maximum_texture1d_mipmapped_width","device__attribute_maximum_texture1d_width","device__attribute_maximum_texture2d_gather_height","device__attribute_maximum_texture2d_gather_width","device__attribute_maximum_texture2d_height","device__attribute_maximum_texture2d_layered_height","device__attribute_maximum_texture2d_layered_layers","device__attribute_maximum_texture2d_layered_width","device__attribute_maximum_texture2d_linear_height","device__attribute_maximum_texture2d_linear_pitch","device__attribute_maximum_texture2d_linear_width","device__attribute_maximum_texture2d_mipmapped_height","device__attribute_maximum_texture2d_mipmapped_width","device__attribute_maximum_texture2d_width","device__attribute_maximum_texture3d_depth","device__attribute_maximum_texture3d_depth_alternate","device__attribute_maximum_texture3d_height","device__attribute_maximum_texture3d_height_alternate","device__attribute_maximum_texture3d_width","device__attribute_maximum_texture3d_width_alternate","device__attribute_maximum_texturecubemap_layered_layers","device__attribute_maximum_texturecubemap_layered_width","device__attribute_maximum_texturecubemap_width","device__attribute_mem_sync_domain_count","device__attribute_memory_clock_rate","device__attribute_memory_pools_supported","device__attribute_mempool_supported_handle_types","device__attribute_mps_enabled","device__attribute_multi_gpu_board","device__attribute_multi_gpu_board_group_id","device__attribute_multicast_supported","device__attribute_multiprocessor_count","device__attribute_num_l2s_per_fbp","device__attribute_num_schedulers_per_multiprocessor","device__attribute_num_tex_per_multiprocessor","device__attribute_numa_config","device__attribute_pageable_memory_access","device__attribute_pageable_memory_access_uses_host_page_tables","device__attribute_pci_bus_id","device__attribute_pci_device_id","device__attribute_pci_domain_id","device__attribute_ram_location","device__attribute_ram_type","device__attribute_reserved_shared_memory_per_block","device__attribute_sass_level","device__attribute_single_to_double_precision_perf_ratio","device__attribute_sparse_cuda_array_supported","device__attribute_stream_priorities_supported","device__attribute_surface_alignment","device__attribute_tcc_driver","device__attribute_tensor_map_access_supported","device__attribute_texture_alignment","device__attribute_texture_pitch_alignment","device__attribute_total_constant_memory","device__attribute_total_memory","device__attribute_unified_addressing","device__attribute_unified_function_pointers","device__attribute_virtual_address_management_supported","device__attribute_warp_size","gpu__compute_memory_throughput.avg.pct_of_peak_sustained_elapsed","gpu__compute_memory_throughput.max.pct_of_peak_sustained_elapsed","gpu__compute_memory_throughput.min.pct_of_peak_sustained_elapsed","gpu__compute_memory_throughput.sum.pct_of_peak_sustained_elapsed","gpu__time_duration.avg","gpu__time_duration.max","gpu__time_duration.min","gpu__time_duration.sum","launch__barrier_count","launch__block_dim_x","launch__block_dim_y","launch__block_dim_z","launch__block_size","launch__cluster_dim_x","launch__cluster_dim_y","launch__cluster_dim_z","launch__cluster_max_active","launch__cluster_max_potential_size","launch__cluster_scheduling_policy","launch__cluster_size","launch__context_id","launch__device_id","launch__func_cache_config","launch__function_pcs","launch__grid_dim_x","launch__grid_dim_y","launch__grid_dim_z","launch__grid_size","launch__kernel_name","launch__occupancy_cluster_gpu_pct","launch__occupancy_cluster_pct","launch__occupancy_limit_barriers","launch__occupancy_limit_blocks","launch__occupancy_limit_registers","launch__occupancy_limit_shared_mem","launch__occupancy_limit_warps","launch__occupancy_per_barrier_count","launch__occupancy_per_block_size","launch__occupancy_per_cluster_size","launch__occupancy_per_register_count","launch__occupancy_per_shared_mem_size","launch__persisting_l2_cache_size","launch__preferred_cluster_dim_x","launch__preferred_cluster_dim_y","launch__preferred_cluster_dim_z","launch__preferred_cluster_size","launch__registers_per_thread","launch__registers_per_thread_allocated","launch__shared_mem_config_size","launch__shared_mem_per_block","launch__shared_mem_per_block_allocated","launch__shared_mem_per_block_driver","launch__shared_mem_per_block_dynamic","launch__shared_mem_per_block_static","launch__sm_count","launch__stack_size","launch__stream_id","launch__thread_count","launch__tpc_count","launch__tpc_enabled","launch__uses_cdp","launch__uses_green_context","launch__uses_mps","launch__uses_nvlink_centric_scheduling","launch__uses_vgpu","launch__waves_per_multiprocessor","lts__t_bytes.avg","lts__t_bytes.max","lts__t_bytes.min","lts__t_bytes.sum","lts__t_sector_hit_rate.pct","numa__cpu_affinity","numa__dev_display_name_all","numa__id_cpu","numa__id_memory","nvlink__bandwidth","nvlink__count_logical","nvlink__count_physical","nvlink__destination_ports","nvlink__dev0Id","nvlink__dev0type","nvlink__dev1Id","nvlink__dev1type","nvlink__dev_display_name_all","nvlink__enabled_mask","nvlink__is_direct_link","nvlink__is_nvswitch_connected","nvlink__max_count","nvlink__peer_access","nvlink__peer_atomic","nvlink__source_ports","nvlink__system_access","nvlink__system_atomic","profiler__perfworks_session_reuse","profiler__replayer_bytes_mem_accessible.avg","profiler__replayer_bytes_mem_accessible.max","profiler__replayer_bytes_mem_accessible.min","profiler__replayer_bytes_mem_accessible.sum","profiler__replayer_bytes_mem_backed_up.avg","profiler__replayer_bytes_mem_backed_up.max","profiler__replayer_bytes_mem_backed_up.min","profiler__replayer_bytes_mem_backed_up.sum","profiler__replayer_passes","profiler__replayer_passes_type_warmup","sm__maximum_warps_avg_per_active_cycle","sm__maximum_warps_per_active_cycle_pct","smsp__maximum_warps_avg_per_active_cycle" +"","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","%","%","%","%","ns","ns","ns","ns","","block","block","block","","","","","cluster","block","","","","","","","","","","","","%","%","block","block","block","block","block","","","","","","byte","","","","","register/thread","register/thread","byte","byte/block","byte/block","byte/block","byte/block","byte/block","SM","","","thread","","","","","","","","","byte","byte","byte","byte","%","","","","","","","","","","","","","","","","","","","","","","","","byte","byte","byte","byte","byte","byte","byte","byte","pass","pass","warp","%","warp" +"0","62","python3.12","127.0.0.1","cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu","1","7","(384, 1, 1)","(296, 168, 1)","0","12.1","0","1","432","1","0","1","1","1","0","1","0","1","0","443","2418000","1","12","1","0","1","1","1","No-CC","1","1","1","0","0","NVIDIA GB10","0","0","256","4","1","1","256","1","0","0","100","1","772935902","11794","32000","16","0","4","2500","16","161","4318","0","1","0","0","1","0","1","443","1","1","1","25165824","24","24","24","1","1","134217728","1024","1024","64","24","2418000","2147483647","65535","65535","4","1","8533000","18874368","2147483647","65536","65536","255","49152","101376","102400","1024","1536","48","12","2048","32768","32768","65536","32768","2048","32768","131072","16384","16384","16384","2046","32768","32768","2048","32768","268435456","32768","131072","32768","32768","65536","32768","2048","32768","65000","2097120","131072","32768","32768","131072","16384","32768","16384","8192","16384","8192","2046","32768","32768","4","8533000","1","1","0","0","0","0","48","6","4","1","0","1","1","1","0","15","1","0","1024","12","64","1","1","512","0","1","512","32","65536","130661769216","1","1","1","32","76.92","76.93","76.90","76.92","26446944","26446944","26446944","26446944","8","384","1","1","384","0","0","0","0","8","PolicySpread","0","1","0","CachePreferNone","1","296","168","1","49728","","0","0","3","24","1","1","4","300","157","0","2028","2388","4718592","0","0","0","0","168","168","102400","89088","89088","1024","88064","0","48","1024","7","19095552","24","all","0","0","0","0","0","1036","2508059386","2508629536","2507690432","40128950176","89.11","1","1","1","1","0","0","0","0","0","0","0","0","1","0","0","0","0","0","0","0","0","0","0","7414590282","7414590282","7414590282","44487541692","7414590282","7414590282","7414590282","44487541692","6","0","12","25","3" +"1","62","python3.12","127.0.0.1","void qk_int_sv_f8_attn_kernel<128, 64, 32, 64, 128, 1, 2, 2, float, 1, __nv_bfloat16, 1, 0, 0, 1, 0, 1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)","1","7","(32, 4, 1)","(296, 56, 1)","0","12.1","0","1","432","1","0","1","1","1","0","1","0","1","0","443","2418000","1","12","1","0","1","1","1","No-CC","1","1","1","0","0","NVIDIA GB10","0","0","256","4","1","1","256","1","0","0","100","1","772935902","11794","32000","16","0","4","2500","16","161","4318","0","1","0","0","1","0","1","443","1","1","1","25165824","24","24","24","1","1","134217728","1024","1024","64","24","2418000","2147483647","65535","65535","4","1","8533000","18874368","2147483647","65536","65536","255","49152","101376","102400","1024","1536","48","12","2048","32768","32768","65536","32768","2048","32768","131072","16384","16384","16384","2046","32768","32768","2048","32768","268435456","32768","131072","32768","32768","65536","32768","2048","32768","65000","2097120","131072","32768","32768","131072","16384","32768","16384","8192","16384","8192","2046","32768","32768","4","8533000","1","1","0","0","0","0","48","6","4","1","0","1","1","1","0","15","1","0","1024","12","64","1","1","512","0","1","512","32","65536","130661769216","1","1","1","32","31.74","31.75","31.74","31.74","257783488","257783488","257783488","257783488","1","32","4","1","128","0","0","0","0","8","PolicySpread","0","1","0","CachePreferNone","1","296","56","1","16576","","0","0","24","24","2","3","12","152","101","0","2732","1200","4718592","0","0","0","0","255","256","102400","33792","33792","1024","32768","0","48","1024","7","2121728","24","all","0","0","0","0","0","172.67","10093623904","10096096896","10092091680","161497982464","98.85","1","1","1","1","0","0","0","0","0","0","0","0","1","0","0","0","0","0","0","0","0","0","0","7414590282","7414590282","7414590282","44487541692","7414590282","7414590282","7414590282","44487541692","6","0","8","16.67","2" +"2","62","python3.12","127.0.0.1","cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu","1","7","(384, 1, 1)","(296, 42, 1)","0","12.1","0","1","432","1","0","1","1","1","0","1","0","1","0","443","2418000","1","12","1","0","1","1","1","No-CC","1","1","1","0","0","NVIDIA GB10","0","0","256","4","1","1","256","1","0","0","100","1","772935902","11794","32000","16","0","4","2500","16","161","4318","0","1","0","0","1","0","1","443","1","1","1","25165824","24","24","24","1","1","134217728","1024","1024","64","24","2418000","2147483647","65535","65535","4","1","8533000","18874368","2147483647","65536","65536","255","49152","101376","102400","1024","1536","48","12","2048","32768","32768","65536","32768","2048","32768","131072","16384","16384","16384","2046","32768","32768","2048","32768","268435456","32768","131072","32768","32768","65536","32768","2048","32768","65000","2097120","131072","32768","32768","131072","16384","32768","16384","8192","16384","8192","2046","32768","32768","4","8533000","1","1","0","0","0","0","48","6","4","1","0","1","1","1","0","15","1","0","1024","12","64","1","1","512","0","1","512","32","65536","130661769216","1","1","1","32","79.51","79.55","79.45","79.51","8440992","8440992","8440992","8440992","8","384","1","1","384","0","0","0","0","8","PolicySpread","0","1","0","CachePreferNone","1","296","42","1","12432","","0","0","3","24","1","1","4","300","157","0","2028","2388","4718592","0","0","0","0","168","168","102400","89088","89088","1024","88064","0","48","1024","7","4773888","24","all","0","0","0","0","0","259","827584636","827995456","826933344","13241354176","89.89","1","1","1","1","0","0","0","0","0","0","0","0","1","0","0","0","0","0","0","0","0","0","0","7414590282","7414590282","7414590282","44487541692","7414590282","7414590282","7414590282","44487541692","6","0","12","25","3" +"3","62","python3.12","127.0.0.1","cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu","1","7","(384, 1, 1)","(296, 224, 1)","0","12.1","0","1","432","1","0","1","1","1","0","1","0","1","0","443","2418000","1","12","1","0","1","1","1","No-CC","1","1","1","0","0","NVIDIA GB10","0","0","256","4","1","1","256","1","0","0","100","1","772935902","11794","32000","16","0","4","2500","16","161","4318","0","1","0","0","1","0","1","443","1","1","1","25165824","24","24","24","1","1","134217728","1024","1024","64","24","2418000","2147483647","65535","65535","4","1","8533000","18874368","2147483647","65536","65536","255","49152","101376","102400","1024","1536","48","12","2048","32768","32768","65536","32768","2048","32768","131072","16384","16384","16384","2046","32768","32768","2048","32768","268435456","32768","131072","32768","32768","65536","32768","2048","32768","65000","2097120","131072","32768","32768","131072","16384","32768","16384","8192","16384","8192","2046","32768","32768","4","8533000","1","1","0","0","0","0","48","6","4","1","0","1","1","1","0","15","1","0","1024","12","64","1","1","512","0","1","512","32","65536","130661769216","1","1","1","32","74.85","74.88","74.83","74.85","36246336","36246336","36246336","36246336","8","384","1","1","384","0","0","0","0","8","PolicySpread","0","1","0","CachePreferNone","1","296","224","1","66304","","0","0","3","24","1","1","4","300","157","0","2028","2388","4718592","0","0","0","0","168","168","102400","89088","89088","1024","88064","0","48","1024","7","25460736","24","all","0","0","0","0","0","1381.33","3344082558","3345239840","3343216480","53505320928","89.11","1","1","1","1","0","0","0","0","0","0","0","0","1","0","0","0","0","0","0","0","0","0","0","7414590282","7414590282","7414590282","44487541692","7414590282","7414590282","7414590282","44487541692","6","0","12","25","3" +"4","62","python3.12","127.0.0.1","cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu","1","7","(384, 1, 1)","(296, 42, 1)","0","12.1","0","1","432","1","0","1","1","1","0","1","0","1","0","443","2418000","1","12","1","0","1","1","1","No-CC","1","1","1","0","0","NVIDIA GB10","0","0","256","4","1","1","256","1","0","0","100","1","772935902","11794","32000","16","0","4","2500","16","161","4318","0","1","0","0","1","0","1","443","1","1","1","25165824","24","24","24","1","1","134217728","1024","1024","64","24","2418000","2147483647","65535","65535","4","1","8533000","18874368","2147483647","65536","65536","255","49152","101376","102400","1024","1536","48","12","2048","32768","32768","65536","32768","2048","32768","131072","16384","16384","16384","2046","32768","32768","2048","32768","268435456","32768","131072","32768","32768","65536","32768","2048","32768","65000","2097120","131072","32768","32768","131072","16384","32768","16384","8192","16384","8192","2046","32768","32768","4","8533000","1","1","0","0","0","0","48","6","4","1","0","1","1","1","0","15","1","0","1024","12","64","1","1","512","0","1","512","32","65536","130661769216","1","1","1","32","77.65","77.66","77.63","77.65","17030400","17030400","17030400","17030400","8","384","1","1","384","0","0","0","0","8","PolicySpread","0","1","0","CachePreferNone","1","296","42","1","12432","","0","0","3","24","1","1","4","300","157","0","2028","2388","4718592","0","0","0","0","168","168","102400","89088","89088","1024","88064","0","48","1024","7","4773888","24","all","0","0","0","0","0","259","1629557740","1629882112","1629239232","26072923840","91.06","1","1","1","1","0","0","0","0","0","0","0","0","1","0","0","0","0","0","0","0","0","0","0","7414590282","7414590282","7414590282","44487541692","7414590282","7414590282","7414590282","44487541692","6","0","12","25","3" diff --git a/benchmarks/gb10-post-fc2-block24-targeted-traffic-20260826.ncu-rep b/benchmarks/gb10-post-fc2-block24-targeted-traffic-20260826.ncu-rep new file mode 100644 index 0000000..72d4c39 Binary files /dev/null and b/benchmarks/gb10-post-fc2-block24-targeted-traffic-20260826.ncu-rep differ diff --git a/benchmarks/gb10-post-fc2-block24-targeted-traffic-capture-20260826.json b/benchmarks/gb10-post-fc2-block24-targeted-traffic-capture-20260826.json new file mode 100644 index 0000000..995fd1b --- /dev/null +++ b/benchmarks/gb10-post-fc2-block24-targeted-traffic-capture-20260826.json @@ -0,0 +1,31 @@ +{ + "block_index": 24, + "hidden_shape": [ + 37810, + 5376 + ], + "segments": [ + [ + 0, + 100, + 1 + ], + [ + 100, + 514, + 2 + ], + [ + 514, + 37810, + 0 + ] + ], + "fused_elementwise": true, + "fused_nvfp4_modulation": true, + "fused_nvfp4_swiglu": true, + "nvfp4_scale_backend": "vortex", + "sage_qkv_layout": "strided_nhd", + "module_forward_checksum": 303055616.0, + "capture": "one warmed block between cudaProfilerStart/Stop" +} \ No newline at end of file diff --git a/benchmarks/gb10-post-fc2-production-profile-summary-20260826.json b/benchmarks/gb10-post-fc2-production-profile-summary-20260826.json new file mode 100644 index 0000000..3b679fc --- /dev/null +++ b/benchmarks/gb10-post-fc2-production-profile-summary-20260826.json @@ -0,0 +1,640 @@ +{ + "name": "gb10-post-fc2-production-profile", + "measurement_date": "2026-08-26", + "source_commit": "a29b8960b0f887c20e74dafa16a24c37d6508b4e", + "measurement_overlay": "the measurement image included deferred step events and latent hashes plus an opt-in 100-token synthetic-text request; the final production overlay gates diagnostics to that benchmark request, and model math is unchanged", + "image": "sha256:a15d0c09dd8cc82aaf2b564d3da760ea5ab8dec974f73075f7d30ac3a504815c", + "device": "NVIDIA GB10", + "compute_capability": "SM121", + "tools": { + "torch": "2.9.1+cu130", + "cuda": "13.0", + "nsight_systems": "2025.3.2.474-253236389321v0", + "nsight_compute": "2025.3.1" + }, + "workload": { + "resolution": [ + 1344, + 768 + ], + "frames": 124, + "packed_tokens": 37810, + "text_tokens": 100, + "steps": 12, + "seed": 440420, + "attention": "sage2" + }, + "configuration": { + "H3_NVFP4_SCALE_BACKEND": "vortex", + "H3_NVFP4_SCALE_VERSION": "1", + "H3_FUSED_ELEMENTWISE": "1", + "H3_NVFP4_MODULATE_FUSION": "1", + "H3_NVFP4_SWIGLU_FUSION": "1", + "H3_NVFP4_FC2_LT_SPLITK1": "1", + "H3_SAGE_QKV_LAYOUT": "strided_nhd" + }, + "authoritative_resident_baseline": { + "sampling_seconds": [ + 256.46369375299946, + 255.44699439899978, + 255.13481797700024 + ], + "median_sampling_seconds": 255.44699439899978, + "per_step_seconds": [ + [ + { + "step": 1, + "seconds": 21.337705078125 + }, + { + "step": 2, + "seconds": 21.40503515625 + }, + { + "step": 3, + "seconds": 21.370228515625 + }, + { + "step": 4, + "seconds": 21.38410546875 + }, + { + "step": 5, + "seconds": 21.38205078125 + }, + { + "step": 6, + "seconds": 21.373046875 + }, + { + "step": 7, + "seconds": 21.376810546875 + }, + { + "step": 8, + "seconds": 21.367021484375 + }, + { + "step": 9, + "seconds": 21.386291015625 + }, + { + "step": 10, + "seconds": 21.36256640625 + }, + { + "step": 11, + "seconds": 21.404072265625 + }, + { + "step": 12, + "seconds": 21.313634765625 + } + ], + [ + { + "step": 1, + "seconds": 21.696716796875 + }, + { + "step": 2, + "seconds": 21.326416015625 + }, + { + "step": 3, + "seconds": 21.3040703125 + }, + { + "step": 4, + "seconds": 21.332494140625 + }, + { + "step": 5, + "seconds": 21.26598046875 + }, + { + "step": 6, + "seconds": 21.267458984375 + }, + { + "step": 7, + "seconds": 21.25258203125 + }, + { + "step": 8, + "seconds": 21.21969140625 + }, + { + "step": 9, + "seconds": 21.242830078125 + }, + { + "step": 10, + "seconds": 21.20158203125 + }, + { + "step": 11, + "seconds": 21.193841796875 + }, + { + "step": 12, + "seconds": 21.141802734375 + } + ], + [ + { + "step": 1, + "seconds": 21.116169921875 + }, + { + "step": 2, + "seconds": 21.237697265625 + }, + { + "step": 3, + "seconds": 21.2424921875 + }, + { + "step": 4, + "seconds": 21.273701171875 + }, + { + "step": 5, + "seconds": 21.275826171875 + }, + { + "step": 6, + "seconds": 21.303318359375 + }, + { + "step": 7, + "seconds": 21.3065546875 + }, + { + "step": 8, + "seconds": 21.32540234375 + }, + { + "step": 9, + "seconds": 21.35030078125 + }, + { + "step": 10, + "seconds": 21.258994140625 + }, + { + "step": 11, + "seconds": 21.249765625 + }, + { + "step": 12, + "seconds": 21.193939453125 + } + ] + ], + "peak_allocated_bytes": [ + 44445830144, + 44445830144, + 44445830144 + ], + "peak_reserved_bytes": [ + 48708452352, + 48708452352, + 48708452352 + ], + "latent_checksums": { + "video_sha256": "c62d23a42972eab907ba42f93c50247ff17a9c454b4a53fe93d2e34f9fefe578", + "audio_sha256": "852005383770480a6503504e1ffec86dd1fb63a69c6400f92da18e39e0986de2", + "exact_across_measured_runs": true + }, + "fc2_dispatch": { + "required_per_run": 600, + "runs": [ + { + "attempts": 600, + "successes": 600, + "fallbacks": 0 + }, + { + "attempts": 600, + "successes": 600, + "fallbacks": 0 + }, + { + "attempts": 600, + "successes": 600, + "fallbacks": 0 + } + ], + "all_passed": true + }, + "measurement_policy": { + "canonical_warmup_runs": 1, + "measured_runs": 3, + "authoritative_timing": "median synchronized resident sampling_seconds", + "step_timing": "deferred CUDA event elapsed time; no per-step synchronization", + "profiling": false + } + }, + "block_24": { + "uninstrumented_module_p50_ms": 427.4100805005219, + "uninstrumented_module_samples": 50, + "synchronized_decomposition_is_attribution_only": true, + "fc2_schedule_note": "profile_h3_block.py decomposes generic NVFP4 linear calls and bypasses forward_swiglu/guarded FC2; use resident NSYS and targeted NCU for production FC2" + }, + "one_warmed_sampling_step_nsys": { + "elapsed_seconds": 20.90812512299999, + "gpu_span_seconds": 20.90717264, + "kernel_time_seconds": 20.895065376, + "gpu_operation_count": 2803, + "kernel_count": 2744, + "kernel_busy_percent_of_span": 99.94209038109325, + "launch_gaps": { + "positive_gap_count": 2743, + "total_ns": 11518912, + "average_ns": 4199.384615384615, + "maximum_ns": 4797728 + }, + "cpu_gpu_overlap": { + "scope": "CUDA kernel-launch API intervals intersected with GPU kernel intervals", + "launch_api_union_ns": 13065552528, + "launch_api_gpu_overlap_ns": 13062771136, + "launch_api_overlap_percent": 99.97871202160002 + }, + "components": { + "sage2": { + "milliseconds": 13029.23856, + "percent_of_kernel_time": 62.355576905566124, + "launches": 250 + }, + "nvfp4_gemms": { + "milliseconds": 4167.855808, + "percent_of_kernel_time": 19.946603339117498, + "launches": 200 + }, + "nvfp4_packing": { + "milliseconds": 2026.853664, + "percent_of_kernel_time": 9.700154689767265, + "launches": 800 + }, + "norm_and_rope": { + "milliseconds": 1084.3856, + "percent_of_kernel_time": 5.1896731619970025, + "launches": 202 + }, + "remaining_gate_add": { + "milliseconds": 549.298592, + "percent_of_kernel_time": 2.6288436150619683, + "launches": 100 + }, + "other": { + "milliseconds": 37.433152, + "percent_of_kernel_time": 0.17914828849014078, + "launches": 1192 + } + } + }, + "block_24_targeted_ncu": { + "source_csv": "benchmarks\\gb10-post-fc2-block24-targeted-20260826.csv", + "source_traffic_csv": "benchmarks\\gb10-post-fc2-block24-targeted-traffic-20260826.csv", + "launch_order_contract": [ + "qkv", + "sage2", + "attention_output", + "fc1", + "fc2" + ], + "cache_control": "none (warmed/uncontrolled cache, as reported by NCU)", + "metrics": { + "qkv": { + "launch_id": 0, + "kernel_name": "cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu", + "grid_size": "(296, 168, 1)", + "block_size": "(384, 1, 1)", + "duration_ns": 26376032.0, + "registers_per_thread": 168.0, + "achieved_occupancy_percent": 20.83, + "eligible_warps_per_scheduler": 0.24, + "issue_active_percent": 18.29, + "sm_throughput_percent": 78.99, + "tensor_pipe_active_percent": 78.99, + "l2_requested_bytes": 40128950176.0, + "l2_hit_rate_percent": 89.11, + "l2_throughput_percent": 77.1, + "memory_throughput_percent": 76.92, + "local_spilling_requests": 0.0, + "top_scheduler_stalls": [ + { + "reason": "sleeping", + "warps_per_issue_active": 3.93 + }, + { + "reason": "math_pipe_throttle", + "warps_per_issue_active": 3.82 + }, + { + "reason": "wait", + "warps_per_issue_active": 3.78 + }, + { + "reason": "selected", + "warps_per_issue_active": 1.0 + }, + { + "reason": "long_scoreboard", + "warps_per_issue_active": 0.61 + } + ] + }, + "sage2": { + "launch_id": 1, + "kernel_name": "void qk_int_sv_f8_attn_kernel<128, 64, 32, 64, 128, 1, 2, 2, float, 1, __nv_bfloat16, 1, 0, 0, 1, 0, 1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)", + "grid_size": "(296, 56, 1)", + "block_size": "(32, 4, 1)", + "duration_ns": 258996000.0, + "registers_per_thread": 255.0, + "achieved_occupancy_percent": 16.65, + "eligible_warps_per_scheduler": 0.46, + "issue_active_percent": 36.49, + "sm_throughput_percent": 75.51, + "tensor_pipe_active_percent": 75.51, + "l2_requested_bytes": 161497982464.0, + "l2_hit_rate_percent": 98.85, + "l2_throughput_percent": 31.59, + "memory_throughput_percent": 31.74, + "local_spilling_requests": 1458688.0, + "top_scheduler_stalls": [ + { + "reason": "wait", + "warps_per_issue_active": 2.01 + }, + { + "reason": "math_pipe_throttle", + "warps_per_issue_active": 1.24 + }, + { + "reason": "selected", + "warps_per_issue_active": 1.0 + }, + { + "reason": "not_selected", + "warps_per_issue_active": 0.26 + }, + { + "reason": "short_scoreboard", + "warps_per_issue_active": 0.25 + } + ] + }, + "attention_output": { + "launch_id": 2, + "kernel_name": "cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu", + "grid_size": "(296, 42, 1)", + "block_size": "(384, 1, 1)", + "duration_ns": 8481152.0, + "registers_per_thread": 168.0, + "achieved_occupancy_percent": 20.83, + "eligible_warps_per_scheduler": 0.23, + "issue_active_percent": 17.51, + "sm_throughput_percent": 81.88, + "tensor_pipe_active_percent": 81.88, + "l2_requested_bytes": 13241354176.0, + "l2_hit_rate_percent": 89.89, + "l2_throughput_percent": 79.1, + "memory_throughput_percent": 79.51, + "local_spilling_requests": 0.0, + "top_scheduler_stalls": [ + { + "reason": "sleeping", + "warps_per_issue_active": 4.08 + }, + { + "reason": "math_pipe_throttle", + "warps_per_issue_active": 3.9 + }, + { + "reason": "wait", + "warps_per_issue_active": 3.85 + }, + { + "reason": "selected", + "warps_per_issue_active": 1.0 + }, + { + "reason": "long_scoreboard", + "warps_per_issue_active": 0.6 + } + ] + }, + "fc1": { + "launch_id": 3, + "kernel_name": "cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu", + "grid_size": "(296, 224, 1)", + "block_size": "(384, 1, 1)", + "duration_ns": 36172608.0, + "registers_per_thread": 168.0, + "achieved_occupancy_percent": 20.83, + "eligible_warps_per_scheduler": 0.23, + "issue_active_percent": 17.39, + "sm_throughput_percent": 76.85, + "tensor_pipe_active_percent": 76.85, + "l2_requested_bytes": 53505320928.0, + "l2_hit_rate_percent": 89.11, + "l2_throughput_percent": 74.99, + "memory_throughput_percent": 74.85, + "local_spilling_requests": 0.0, + "top_scheduler_stalls": [ + { + "reason": "sleeping", + "warps_per_issue_active": 4.41 + }, + { + "reason": "math_pipe_throttle", + "warps_per_issue_active": 3.82 + }, + { + "reason": "wait", + "warps_per_issue_active": 3.78 + }, + { + "reason": "selected", + "warps_per_issue_active": 1.0 + }, + { + "reason": "long_scoreboard", + "warps_per_issue_active": 0.61 + } + ] + }, + "fc2": { + "launch_id": 4, + "kernel_name": "cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu", + "grid_size": "(296, 42, 1)", + "block_size": "(384, 1, 1)", + "duration_ns": 16198976.0, + "registers_per_thread": 168.0, + "achieved_occupancy_percent": 20.83, + "eligible_warps_per_scheduler": 0.23, + "issue_active_percent": 17.44, + "sm_throughput_percent": 85.88, + "tensor_pipe_active_percent": 85.88, + "l2_requested_bytes": 26072923840.0, + "l2_hit_rate_percent": 91.06, + "l2_throughput_percent": 81.58, + "memory_throughput_percent": 77.65, + "local_spilling_requests": 0.0, + "top_scheduler_stalls": [ + { + "reason": "math_pipe_throttle", + "warps_per_issue_active": 4.06 + }, + { + "reason": "wait", + "warps_per_issue_active": 3.98 + }, + { + "reason": "sleeping", + "warps_per_issue_active": 3.81 + }, + { + "reason": "selected", + "warps_per_issue_active": 1.0 + }, + { + "reason": "long_scoreboard", + "warps_per_issue_active": 0.6 + } + ] + } + }, + "units": { + "duration_ns": "ns", + "l2_requested_bytes": "lts__t_bytes.sum", + "throughput_and_hit_rate": "%" + } + }, + "comparison_to_pre_fc2_profile": { + "previous_summary": "benchmarks/gb10-fully-fused-fresh-nsight-summary.json", + "block_24_p50_ms": { + "previous": 458.7751985236537, + "current": 427.4100805005219, + "absolute_change": -31.365118023131856, + "percent_change": -6.83670741663136 + }, + "block_24_comparison_note": "same-script decomposition control only; it excludes guarded FC2 and must not be interpreted as the production FC2 gain", + "warmed_step_gpu_span_seconds": { + "previous": 23.706977664, + "current": 20.90717264, + "absolute_change": -2.799805024000001, + "percent_change": -11.810046239051452 + }, + "warmed_step_kernel_time_seconds": { + "previous": 23.693982688, + "current": 20.895065376, + "absolute_change": -2.798917311999997, + "percent_change": -11.812776893002164 + }, + "warmed_step_kernel_count": { + "previous": 2694, + "current": 2744, + "absolute_change": 50, + "percent_change": 1.8559762435040872 + }, + "components": { + "sage2": { + "previous": 13760.566304, + "current": 13029.23856, + "absolute_change": -731.3277440000002, + "percent_change": -5.314663131178065 + }, + "nvfp4_gemms": { + "previous": 6269.804512, + "current": 4167.855808, + "absolute_change": -2101.9487039999995, + "percent_change": -33.52494802632212 + }, + "nvfp4_packing": { + "previous": 2093.052672, + "current": 2026.853664, + "absolute_change": -66.19900799999982, + "percent_change": -3.1627970421185814 + }, + "norm_and_rope": { + "previous": 987.609696, + "current": 1084.3856, + "absolute_change": 96.77590400000008, + "percent_change": 9.79900302639396 + }, + "remaining_gate_add": { + "previous": 545.150976, + "current": 549.298592, + "absolute_change": 4.147615999999971, + "percent_change": 0.7608196963037273 + }, + "other": { + "previous": 37.798528, + "current": 37.433152, + "absolute_change": -0.3653759999999977, + "percent_change": -0.966640817335529 + } + } + }, + "bottleneck_ranking": [ + { + "component": "sage2", + "milliseconds": 13029.23856, + "percent_of_kernel_time": 62.355576905566124 + }, + { + "component": "nvfp4_gemms", + "milliseconds": 4167.855808, + "percent_of_kernel_time": 19.946603339117498 + }, + { + "component": "nvfp4_packing", + "milliseconds": 2026.853664, + "percent_of_kernel_time": 9.700154689767265 + }, + { + "component": "norm_and_rope", + "milliseconds": 1084.3856, + "percent_of_kernel_time": 5.1896731619970025 + }, + { + "component": "remaining_gate_add", + "milliseconds": 549.298592, + "percent_of_kernel_time": 2.6288436150619683 + }, + { + "component": "other", + "milliseconds": 37.433152, + "percent_of_kernel_time": 0.17914828849014078 + } + ], + "decision": { + "authoritative_exact_baseline": "255.446994 s median resident sampling", + "time_bottleneck": "Sage2 remains dominant at 62.36% of warmed-step kernel time; its mainloop is 241.91 ms average in NSYS and 259.00 ms in the NCU replay.", + "secondary_bottlenecks": "NVFP4 GEMMs are 19.95%, packing 9.70%, norm/RoPE 5.19%, and gate/add 2.63%.", + "next_optimization": "None started. Sage2 is the next-ranked investigation target; any implementation requires a separate approved experiment after this baseline is accepted." + }, + "artifacts": [ + "benchmarks/gb10-post-fc2-resident-baseline-20260826.json", + "benchmarks/gb10-post-fc2-block24-profile-20260826.json", + "benchmarks/gb10-post-fc2-warmed-step-capture-20260826.json", + "benchmarks/gb10-post-fc2-warmed-step-20260826.nsys-rep", + "benchmarks/gb10-post-fc2-warmed-step-20260826.sqlite", + "benchmarks/gb10-post-fc2-warmed-step-stats_cuda_gpu_trace.csv", + "benchmarks/gb10-post-fc2-warmed-step-stats_cuda_kern_exec_trace.csv", + "benchmarks/gb10-post-fc2-warmed-step-stats_nvtx_gpu_proj_trace.csv", + "benchmarks/gb10-post-fc2-warmed-step-stats_cuda_gpu_kern_sum.csv", + "benchmarks/gb10-post-fc2-warmed-step-stats_cuda_api_sum.csv", + "benchmarks/gb10-post-fc2-block24-targeted-capture-20260826.json", + "benchmarks/gb10-post-fc2-block24-targeted-20260826.ncu-rep", + "benchmarks/gb10-post-fc2-block24-targeted-20260826.csv", + "benchmarks/gb10-post-fc2-block24-targeted-traffic-capture-20260826.json", + "benchmarks/gb10-post-fc2-block24-targeted-traffic-20260826.ncu-rep", + "benchmarks/gb10-post-fc2-block24-targeted-traffic-20260826.csv", + "benchmarks/gb10-post-fc2-warmed-step-nsys-summary-20260826.json", + "benchmarks/gb10-post-fc2-block24-targeted-summary-20260826.json" + ] +} diff --git a/benchmarks/gb10-post-fc2-resident-baseline-20260826.json b/benchmarks/gb10-post-fc2-resident-baseline-20260826.json new file mode 100644 index 0000000..4900bd9 --- /dev/null +++ b/benchmarks/gb10-post-fc2-resident-baseline-20260826.json @@ -0,0 +1,1140 @@ +{ + "name": "gb10-post-fc2-resident-baseline", + "source_commit": "a29b8960b0f887c20e74dafa16a24c37d6508b4e", + "image": "sha256:a15d0c09dd8cc82aaf2b564d3da760ea5ab8dec974f73075f7d30ac3a504815c", + "measurement_date": "2026-08-26", + "workload": { + "resolution": [ + 1344, + 768 + ], + "frames": 124, + "steps": 12, + "seed": 440420, + "attention": "sage2", + "prompt": "A playful orange tabby cat starts in an ordinary cozy living room in a normal house, afternoon light, sofa and rug. The cat crouches, jumps, and does one clean athletic backflip in slow motion. As the backflip completes there is a sharp cinematic cut: the cat lands perfectly on a glowing neon disco dance floor wearing oversized black sunglasses. Mirror ball reflections, colorful lights, joyful party energy, stylish and funny, clear before-and-after transformation.", + "prompt_sha256": "c7454f25e4df66fa05f5bad5f612e3532bf1e58d80dbb9b139493822f14d0988" + }, + "measurement_policy": { + "canonical_warmup_runs": 1, + "measured_runs": 3, + "authoritative_timing": "median synchronized resident sampling_seconds", + "step_timing": "deferred CUDA event elapsed time; no per-step synchronization", + "profiling": false + }, + "sampling_seconds": [ + 256.46369375299946, + 255.44699439899978, + 255.13481797700024 + ], + "median_sampling_seconds": 255.44699439899978, + "sampling_steps": [ + [ + { + "step": 1, + "seconds": 21.337705078125 + }, + { + "step": 2, + "seconds": 21.40503515625 + }, + { + "step": 3, + "seconds": 21.370228515625 + }, + { + "step": 4, + "seconds": 21.38410546875 + }, + { + "step": 5, + "seconds": 21.38205078125 + }, + { + "step": 6, + "seconds": 21.373046875 + }, + { + "step": 7, + "seconds": 21.376810546875 + }, + { + "step": 8, + "seconds": 21.367021484375 + }, + { + "step": 9, + "seconds": 21.386291015625 + }, + { + "step": 10, + "seconds": 21.36256640625 + }, + { + "step": 11, + "seconds": 21.404072265625 + }, + { + "step": 12, + "seconds": 21.313634765625 + } + ], + [ + { + "step": 1, + "seconds": 21.696716796875 + }, + { + "step": 2, + "seconds": 21.326416015625 + }, + { + "step": 3, + "seconds": 21.3040703125 + }, + { + "step": 4, + "seconds": 21.332494140625 + }, + { + "step": 5, + "seconds": 21.26598046875 + }, + { + "step": 6, + "seconds": 21.267458984375 + }, + { + "step": 7, + "seconds": 21.25258203125 + }, + { + "step": 8, + "seconds": 21.21969140625 + }, + { + "step": 9, + "seconds": 21.242830078125 + }, + { + "step": 10, + "seconds": 21.20158203125 + }, + { + "step": 11, + "seconds": 21.193841796875 + }, + { + "step": 12, + "seconds": 21.141802734375 + } + ], + [ + { + "step": 1, + "seconds": 21.116169921875 + }, + { + "step": 2, + "seconds": 21.237697265625 + }, + { + "step": 3, + "seconds": 21.2424921875 + }, + { + "step": 4, + "seconds": 21.273701171875 + }, + { + "step": 5, + "seconds": 21.275826171875 + }, + { + "step": 6, + "seconds": 21.303318359375 + }, + { + "step": 7, + "seconds": 21.3065546875 + }, + { + "step": 8, + "seconds": 21.32540234375 + }, + { + "step": 9, + "seconds": 21.35030078125 + }, + { + "step": 10, + "seconds": 21.258994140625 + }, + { + "step": 11, + "seconds": 21.249765625 + }, + { + "step": 12, + "seconds": 21.193939453125 + } + ] + ], + "sampling_peak_allocated_bytes": [ + 44445830144, + 44445830144, + 44445830144 + ], + "sampling_peak_reserved_bytes": [ + 48708452352, + 48708452352, + 48708452352 + ], + "latent_checksums": { + "video_sha256": "c62d23a42972eab907ba42f93c50247ff17a9c454b4a53fe93d2e34f9fefe578", + "audio_sha256": "852005383770480a6503504e1ffec86dd1fb63a69c6400f92da18e39e0986de2", + "exact_across_measured_runs": true + }, + "fc2_dispatch": { + "required_per_run": 600, + "runs": [ + { + "attempts": 600, + "successes": 600, + "fallbacks": 0 + }, + { + "attempts": 600, + "successes": 600, + "fallbacks": 0 + }, + { + "attempts": 600, + "successes": 600, + "fallbacks": 0 + } + ], + "all_passed": true + }, + "canonical_warmup": { + "output": "/output/h3-blackwell-runtime/post-fc2-canonical-warmup.mp4", + "audio_output": null, + "frames": 124, + "width": 1344, + "height": 768, + "source_width": 1344, + "source_height": 768, + "seed": 440420, + "attention": "sage2", + "turbo": null, + "upscale": null, + "keep_intermediates": false, + "vae_dtype": "float16", + "vae_tile_size": 256, + "benchmark_text_tokens": 100, + "sampling_seconds": 254.57577561699964, + "sampling_steps": [ + { + "step": 1, + "seconds": 21.979201171875 + }, + { + "step": 2, + "seconds": 20.929982421875 + }, + { + "step": 3, + "seconds": 21.044685546875 + }, + { + "step": 4, + "seconds": 21.101681640625 + }, + { + "step": 5, + "seconds": 21.140072265625 + }, + { + "step": 6, + "seconds": 21.139208984375 + }, + { + "step": 7, + "seconds": 21.20319140625 + }, + { + "step": 8, + "seconds": 21.186095703125 + }, + { + "step": 9, + "seconds": 21.224865234375 + }, + { + "step": 10, + "seconds": 21.193998046875 + }, + { + "step": 11, + "seconds": 21.234625 + }, + { + "step": 12, + "seconds": 21.1969296875 + } + ], + "sampling_peak_allocated_bytes": 44445830144, + "sampling_peak_reserved_bytes": 47171239936, + "latent_checksums": { + "video_sha256": "c62d23a42972eab907ba42f93c50247ff17a9c454b4a53fe93d2e34f9fefe578", + "audio_sha256": "852005383770480a6503504e1ffec86dd1fb63a69c6400f92da18e39e0986de2" + }, + "fc2_dispatch_delta": { + "attempts": 600, + "successes": 600, + "fallbacks": 0 + }, + "stages": [ + { + "stage": "latents_initialized", + "seconds": 0.0294893440004671 + }, + { + "stage": "text_conditioned", + "seconds": 0.0036787760000152048 + }, + { + "stage": "sampled", + "seconds": 254.57577561699964 + }, + { + "stage": "vae_decoded", + "seconds": 39.0833450199998 + }, + { + "stage": "pixels_cpu", + "seconds": 0.47611793799933366 + }, + { + "stage": "raw_write", + "seconds": 5.7348896350004 + }, + { + "stage": "video_encode", + "seconds": 0.8170129330001146 + }, + { + "stage": "audio_decoded", + "seconds": 0.3539261100004296 + }, + { + "stage": "audio_raw_write", + "seconds": 0.01070119700034411 + }, + { + "stage": "audio_encode", + "seconds": 0.03721532300005492 + }, + { + "stage": "mux", + "seconds": 0.21110165600020991 + } + ], + "cache": { + "mode": null, + "threshold": 0.0, + "skipped_steps": 0, + "rates": [] + }, + "request_seconds": 301.3332535490008, + "wall_seconds": 301.4555453029998 + }, + "measured_responses": [ + { + "output": "/output/h3-blackwell-runtime/post-fc2-canonical-run-1.mp4", + "audio_output": null, + "frames": 124, + "width": 1344, + "height": 768, + "source_width": 1344, + "source_height": 768, + "seed": 440420, + "attention": "sage2", + "turbo": null, + "upscale": null, + "keep_intermediates": false, + "vae_dtype": "float16", + "vae_tile_size": 256, + "benchmark_text_tokens": 100, + "sampling_seconds": 256.46369375299946, + "sampling_steps": [ + { + "step": 1, + "seconds": 21.337705078125 + }, + { + "step": 2, + "seconds": 21.40503515625 + }, + { + "step": 3, + "seconds": 21.370228515625 + }, + { + "step": 4, + "seconds": 21.38410546875 + }, + { + "step": 5, + "seconds": 21.38205078125 + }, + { + "step": 6, + "seconds": 21.373046875 + }, + { + "step": 7, + "seconds": 21.376810546875 + }, + { + "step": 8, + "seconds": 21.367021484375 + }, + { + "step": 9, + "seconds": 21.386291015625 + }, + { + "step": 10, + "seconds": 21.36256640625 + }, + { + "step": 11, + "seconds": 21.404072265625 + }, + { + "step": 12, + "seconds": 21.313634765625 + } + ], + "sampling_peak_allocated_bytes": 44445830144, + "sampling_peak_reserved_bytes": 48708452352, + "latent_checksums": { + "video_sha256": "c62d23a42972eab907ba42f93c50247ff17a9c454b4a53fe93d2e34f9fefe578", + "audio_sha256": "852005383770480a6503504e1ffec86dd1fb63a69c6400f92da18e39e0986de2" + }, + "fc2_dispatch_delta": { + "attempts": 600, + "successes": 600, + "fallbacks": 0 + }, + "stages": [ + { + "stage": "latents_initialized", + "seconds": 0.030945731999963755 + }, + { + "stage": "text_conditioned", + "seconds": 5.3199999456410296e-05 + }, + { + "stage": "sampled", + "seconds": 256.46369375299946 + }, + { + "stage": "vae_decoded", + "seconds": 39.04563350299941 + }, + { + "stage": "pixels_cpu", + "seconds": 0.13809898999988945 + }, + { + "stage": "raw_write", + "seconds": 5.884286710000197 + }, + { + "stage": "video_encode", + "seconds": 0.8152395229999456 + }, + { + "stage": "audio_decoded", + "seconds": 0.2907312010001988 + }, + { + "stage": "audio_raw_write", + "seconds": 0.011123714000859763 + }, + { + "stage": "audio_encode", + "seconds": 0.035295738999593596 + }, + { + "stage": "mux", + "seconds": 0.22826891600016097 + } + ], + "cache": { + "mode": null, + "threshold": 0.0, + "skipped_steps": 0, + "rates": [] + }, + "request_seconds": 302.94337098099913, + "wall_seconds": 303.06157443899974, + "client_wall_seconds": 303.0632682179994 + }, + { + "output": "/output/h3-blackwell-runtime/post-fc2-canonical-run-2.mp4", + "audio_output": null, + "frames": 124, + "width": 1344, + "height": 768, + "source_width": 1344, + "source_height": 768, + "seed": 440420, + "attention": "sage2", + "turbo": null, + "upscale": null, + "keep_intermediates": false, + "vae_dtype": "float16", + "vae_tile_size": 256, + "benchmark_text_tokens": 100, + "sampling_seconds": 255.44699439899978, + "sampling_steps": [ + { + "step": 1, + "seconds": 21.696716796875 + }, + { + "step": 2, + "seconds": 21.326416015625 + }, + { + "step": 3, + "seconds": 21.3040703125 + }, + { + "step": 4, + "seconds": 21.332494140625 + }, + { + "step": 5, + "seconds": 21.26598046875 + }, + { + "step": 6, + "seconds": 21.267458984375 + }, + { + "step": 7, + "seconds": 21.25258203125 + }, + { + "step": 8, + "seconds": 21.21969140625 + }, + { + "step": 9, + "seconds": 21.242830078125 + }, + { + "step": 10, + "seconds": 21.20158203125 + }, + { + "step": 11, + "seconds": 21.193841796875 + }, + { + "step": 12, + "seconds": 21.141802734375 + } + ], + "sampling_peak_allocated_bytes": 44445830144, + "sampling_peak_reserved_bytes": 48708452352, + "latent_checksums": { + "video_sha256": "c62d23a42972eab907ba42f93c50247ff17a9c454b4a53fe93d2e34f9fefe578", + "audio_sha256": "852005383770480a6503504e1ffec86dd1fb63a69c6400f92da18e39e0986de2" + }, + "fc2_dispatch_delta": { + "attempts": 600, + "successes": 600, + "fallbacks": 0 + }, + "stages": [ + { + "stage": "latents_initialized", + "seconds": 0.0244528100001844 + }, + { + "stage": "text_conditioned", + "seconds": 5.4271999943011906e-05 + }, + { + "stage": "sampled", + "seconds": 255.44699439899978 + }, + { + "stage": "vae_decoded", + "seconds": 38.825791856999786 + }, + { + "stage": "pixels_cpu", + "seconds": 0.14175554799930978 + }, + { + "stage": "raw_write", + "seconds": 5.936753373000101 + }, + { + "stage": "video_encode", + "seconds": 0.8132764319998387 + }, + { + "stage": "audio_decoded", + "seconds": 0.29106337200028065 + }, + { + "stage": "audio_raw_write", + "seconds": 0.011491439000565151 + }, + { + "stage": "audio_encode", + "seconds": 0.036401267000655935 + }, + { + "stage": "mux", + "seconds": 0.21376025900008244 + } + ], + "cache": { + "mode": null, + "threshold": 0.0, + "skipped_steps": 0, + "rates": [] + }, + "request_seconds": 301.7417950280005, + "wall_seconds": 301.86345436899956, + "client_wall_seconds": 301.866534363 + }, + { + "output": "/output/h3-blackwell-runtime/post-fc2-canonical-run-3.mp4", + "audio_output": null, + "frames": 124, + "width": 1344, + "height": 768, + "source_width": 1344, + "source_height": 768, + "seed": 440420, + "attention": "sage2", + "turbo": null, + "upscale": null, + "keep_intermediates": false, + "vae_dtype": "float16", + "vae_tile_size": 256, + "benchmark_text_tokens": 100, + "sampling_seconds": 255.13481797700024, + "sampling_steps": [ + { + "step": 1, + "seconds": 21.116169921875 + }, + { + "step": 2, + "seconds": 21.237697265625 + }, + { + "step": 3, + "seconds": 21.2424921875 + }, + { + "step": 4, + "seconds": 21.273701171875 + }, + { + "step": 5, + "seconds": 21.275826171875 + }, + { + "step": 6, + "seconds": 21.303318359375 + }, + { + "step": 7, + "seconds": 21.3065546875 + }, + { + "step": 8, + "seconds": 21.32540234375 + }, + { + "step": 9, + "seconds": 21.35030078125 + }, + { + "step": 10, + "seconds": 21.258994140625 + }, + { + "step": 11, + "seconds": 21.249765625 + }, + { + "step": 12, + "seconds": 21.193939453125 + } + ], + "sampling_peak_allocated_bytes": 44445830144, + "sampling_peak_reserved_bytes": 48708452352, + "latent_checksums": { + "video_sha256": "c62d23a42972eab907ba42f93c50247ff17a9c454b4a53fe93d2e34f9fefe578", + "audio_sha256": "852005383770480a6503504e1ffec86dd1fb63a69c6400f92da18e39e0986de2" + }, + "fc2_dispatch_delta": { + "attempts": 600, + "successes": 600, + "fallbacks": 0 + }, + "stages": [ + { + "stage": "latents_initialized", + "seconds": 0.02409168200028944 + }, + { + "stage": "text_conditioned", + "seconds": 5.864000013389159e-05 + }, + { + "stage": "sampled", + "seconds": 255.13481797700024 + }, + { + "stage": "vae_decoded", + "seconds": 38.90338989799966 + }, + { + "stage": "pixels_cpu", + "seconds": 0.13844444200003636 + }, + { + "stage": "raw_write", + "seconds": 5.898542812999949 + }, + { + "stage": "video_encode", + "seconds": 0.8105222340000182 + }, + { + "stage": "audio_decoded", + "seconds": 0.2911434189991269 + }, + { + "stage": "audio_raw_write", + "seconds": 0.0007821060007699998 + }, + { + "stage": "audio_encode", + "seconds": 0.037410761000501225 + }, + { + "stage": "mux", + "seconds": 0.2268666479994863 + } + ], + "cache": { + "mode": null, + "threshold": 0.0, + "skipped_steps": 0, + "rates": [] + }, + "request_seconds": 301.4660706200002, + "wall_seconds": 301.6021307559995, + "client_wall_seconds": 301.6040240329994 + } + ], + "ready_before": { + "ready": true, + "attention_backends": [ + "sage2", + "cudnn_sdpa", + "ck_int8", + "sdpa", + "flash4", + "sage3", + "sage3_mean", + "kj_sage_cuda", + "kj_sage_triton", + "kj_sage_fp8", + "kj_sage_fp8pp", + "kj_head_sliced", + "sol_attn" + ], + "attention_backend_status": { + "sage2": "available", + "cudnn_sdpa": "available: forced cuDNN SDPA with no backend fallback", + "ck_int8": "available: approximate Comfy Kitchen INT8 Q/K/V attention", + "sdpa": "available", + "flash4": "available: official FlashAttention-4 CuTeDSL Blackwell kernel (strict, no fallback)", + "sage3": "available", + "sage3_mean": "available", + "kj_sage_cuda": "available", + "kj_sage_triton": "available", + "kj_sage_fp8": "available", + "kj_sage_fp8pp": "available", + "kj_head_sliced": "available", + "sol_attn": "experimental: sparse Triton attention for eligible non-causal H3 attention calls; falls back below H3_SOL_MIN_TOKENS unless H3_SOL_STRICT=1", + "easycache": "planned: approximate denoiser cache", + "h3_cache": "planned: approximate H3-specific cache", + "kj_chunked_ffn": "available: exact H3 MLP row chunking via H3_MLP_CHUNKS or runtime args" + }, + "runtime": { + "ready": true, + "initial_attention": "sage2", + "current_attention": "sage2", + "available_turbos": [ + "4step", + "8step" + ], + "current_turbo": null, + "latent_upscaler_loaded": true, + "vae_dtype": "float16", + "vae_tile_size": 256, + "mlp_chunks": 1, + "mlp_chunk_threshold": 4096, + "fc2_lt": { + "enabled": true, + "attempts": 0, + "successes": 0, + "fallbacks": 100 + }, + "loaded_at": 1787682891.127061, + "load_stages": [ + { + "stage": "qwen_loaded", + "seconds": 23.215869685999678 + }, + { + "stage": "h3_loaded", + "seconds": 16.684102700999574 + }, + { + "stage": "token_refiner_loaded", + "seconds": 0.0003298859992355574 + }, + { + "stage": "turbo_4step_loaded", + "seconds": 15.194392286000038 + }, + { + "stage": "turbo_8step_loaded", + "seconds": 15.133038609000323 + }, + { + "stage": "video_vae_loaded", + "seconds": 6.184747135999714 + }, + { + "stage": "audio_vae_loaded", + "seconds": 2.1314666650005165 + }, + { + "stage": "vae_encoder_loaded", + "seconds": 2.953530530000535 + }, + { + "stage": "vision_tower_loaded", + "seconds": 7.438691268000184 + }, + { + "stage": "latent_upscaler_loaded", + "seconds": 1.0793168410000362 + } + ] + }, + "warmup_result": { + "output": "/output/h3-blackwell-runtime/hot-runtime-warmup.mp4", + "audio_output": null, + "frames": 22, + "width": 320, + "height": 192, + "source_width": 320, + "source_height": 192, + "seed": 440501, + "attention": "sage2", + "turbo": null, + "upscale": null, + "keep_intermediates": false, + "vae_dtype": "float16", + "vae_tile_size": 256, + "benchmark_text_tokens": null, + "sampling_seconds": 25.51713720600037, + "sampling_steps": [ + { + "step": 1, + "seconds": 25.33962109375 + }, + { + "step": 2, + "seconds": 0.17677439880371093 + } + ], + "sampling_peak_allocated_bytes": 40093123584, + "sampling_peak_reserved_bytes": 40932212736, + "latent_checksums": { + "video_sha256": "cd8067e87877b0795b74cb0a2b424369ecdb36f38745957d9e42b4d775fd68ee", + "audio_sha256": "a133fa0a5639e6cb51cf6231360ad9050190410e08acf542fb7c2d5f2d3984df" + }, + "fc2_dispatch_delta": { + "attempts": 0, + "successes": 0, + "fallbacks": 100 + }, + "stages": [ + { + "stage": "latents_initialized", + "seconds": 0.016533465999600594 + }, + { + "stage": "text_conditioned", + "seconds": 2.3074946679998902 + }, + { + "stage": "sampled", + "seconds": 25.51713720600037 + }, + { + "stage": "vae_decoded", + "seconds": 1.394877304000147 + }, + { + "stage": "pixels_cpu", + "seconds": 0.0005727660000047763 + }, + { + "stage": "raw_write", + "seconds": 0.06253001799996127 + }, + { + "stage": "video_encode", + "seconds": 0.31031726600031106 + }, + { + "stage": "audio_decoded", + "seconds": 0.052017392000379914 + }, + { + "stage": "audio_raw_write", + "seconds": 0.0008239339995270711 + }, + { + "stage": "audio_encode", + "seconds": 0.03522011000040948 + }, + { + "stage": "mux", + "seconds": 0.07208253599947057 + } + ], + "cache": { + "mode": null, + "threshold": 0.0, + "skipped_steps": 0, + "rates": [] + }, + "request_seconds": 29.769606666000072 + } + }, + "ready_after": { + "ready": true, + "attention_backends": [ + "sage2", + "cudnn_sdpa", + "ck_int8", + "sdpa", + "flash4", + "sage3", + "sage3_mean", + "kj_sage_cuda", + "kj_sage_triton", + "kj_sage_fp8", + "kj_sage_fp8pp", + "kj_head_sliced", + "sol_attn" + ], + "attention_backend_status": { + "sage2": "available", + "cudnn_sdpa": "available: forced cuDNN SDPA with no backend fallback", + "ck_int8": "available: approximate Comfy Kitchen INT8 Q/K/V attention", + "sdpa": "available", + "flash4": "available: official FlashAttention-4 CuTeDSL Blackwell kernel (strict, no fallback)", + "sage3": "available", + "sage3_mean": "available", + "kj_sage_cuda": "available", + "kj_sage_triton": "available", + "kj_sage_fp8": "available", + "kj_sage_fp8pp": "available", + "kj_head_sliced": "available", + "sol_attn": "experimental: sparse Triton attention for eligible non-causal H3 attention calls; falls back below H3_SOL_MIN_TOKENS unless H3_SOL_STRICT=1", + "easycache": "planned: approximate denoiser cache", + "h3_cache": "planned: approximate H3-specific cache", + "kj_chunked_ffn": "available: exact H3 MLP row chunking via H3_MLP_CHUNKS or runtime args" + }, + "runtime": { + "ready": true, + "initial_attention": "sage2", + "current_attention": "sage2", + "available_turbos": [ + "4step", + "8step" + ], + "current_turbo": null, + "latent_upscaler_loaded": true, + "vae_dtype": "float16", + "vae_tile_size": 256, + "mlp_chunks": 1, + "mlp_chunk_threshold": 4096, + "fc2_lt": { + "enabled": true, + "attempts": 2400, + "successes": 2400, + "fallbacks": 100 + }, + "loaded_at": 1787682891.127061, + "load_stages": [ + { + "stage": "qwen_loaded", + "seconds": 23.215869685999678 + }, + { + "stage": "h3_loaded", + "seconds": 16.684102700999574 + }, + { + "stage": "token_refiner_loaded", + "seconds": 0.0003298859992355574 + }, + { + "stage": "turbo_4step_loaded", + "seconds": 15.194392286000038 + }, + { + "stage": "turbo_8step_loaded", + "seconds": 15.133038609000323 + }, + { + "stage": "video_vae_loaded", + "seconds": 6.184747135999714 + }, + { + "stage": "audio_vae_loaded", + "seconds": 2.1314666650005165 + }, + { + "stage": "vae_encoder_loaded", + "seconds": 2.953530530000535 + }, + { + "stage": "vision_tower_loaded", + "seconds": 7.438691268000184 + }, + { + "stage": "latent_upscaler_loaded", + "seconds": 1.0793168410000362 + } + ] + }, + "warmup_result": { + "output": "/output/h3-blackwell-runtime/hot-runtime-warmup.mp4", + "audio_output": null, + "frames": 22, + "width": 320, + "height": 192, + "source_width": 320, + "source_height": 192, + "seed": 440501, + "attention": "sage2", + "turbo": null, + "upscale": null, + "keep_intermediates": false, + "vae_dtype": "float16", + "vae_tile_size": 256, + "benchmark_text_tokens": null, + "sampling_seconds": 25.51713720600037, + "sampling_steps": [ + { + "step": 1, + "seconds": 25.33962109375 + }, + { + "step": 2, + "seconds": 0.17677439880371093 + } + ], + "sampling_peak_allocated_bytes": 40093123584, + "sampling_peak_reserved_bytes": 40932212736, + "latent_checksums": { + "video_sha256": "cd8067e87877b0795b74cb0a2b424369ecdb36f38745957d9e42b4d775fd68ee", + "audio_sha256": "a133fa0a5639e6cb51cf6231360ad9050190410e08acf542fb7c2d5f2d3984df" + }, + "fc2_dispatch_delta": { + "attempts": 0, + "successes": 0, + "fallbacks": 100 + }, + "stages": [ + { + "stage": "latents_initialized", + "seconds": 0.016533465999600594 + }, + { + "stage": "text_conditioned", + "seconds": 2.3074946679998902 + }, + { + "stage": "sampled", + "seconds": 25.51713720600037 + }, + { + "stage": "vae_decoded", + "seconds": 1.394877304000147 + }, + { + "stage": "pixels_cpu", + "seconds": 0.0005727660000047763 + }, + { + "stage": "raw_write", + "seconds": 0.06253001799996127 + }, + { + "stage": "video_encode", + "seconds": 0.31031726600031106 + }, + { + "stage": "audio_decoded", + "seconds": 0.052017392000379914 + }, + { + "stage": "audio_raw_write", + "seconds": 0.0008239339995270711 + }, + { + "stage": "audio_encode", + "seconds": 0.03522011000040948 + }, + { + "stage": "mux", + "seconds": 0.07208253599947057 + } + ], + "cache": { + "mode": null, + "threshold": 0.0, + "skipped_steps": 0, + "rates": [] + }, + "request_seconds": 29.769606666000072 + } + } +} diff --git a/benchmarks/gb10-post-fc2-warmed-step-20260826.nsys-rep b/benchmarks/gb10-post-fc2-warmed-step-20260826.nsys-rep new file mode 100644 index 0000000..ef0cde2 Binary files /dev/null and b/benchmarks/gb10-post-fc2-warmed-step-20260826.nsys-rep differ diff --git a/benchmarks/gb10-post-fc2-warmed-step-20260826.sqlite b/benchmarks/gb10-post-fc2-warmed-step-20260826.sqlite new file mode 100644 index 0000000..672eacc Binary files /dev/null and b/benchmarks/gb10-post-fc2-warmed-step-20260826.sqlite differ diff --git a/benchmarks/gb10-post-fc2-warmed-step-capture-20260826.json b/benchmarks/gb10-post-fc2-warmed-step-capture-20260826.json new file mode 100644 index 0000000..6fb00a0 --- /dev/null +++ b/benchmarks/gb10-post-fc2-warmed-step-capture-20260826.json @@ -0,0 +1,67 @@ +{ + "device": "NVIDIA GB10", + "torch": "2.9.1+cu130", + "attention": "sage2", + "resolution": [ + 1344, + 768 + ], + "frames": 124, + "steps": 1, + "seed": 440420, + "text_tokens": 100, + "warmup_runs": 1, + "cuda_profiler_capture": true, + "argv": [ + "tools/profile_sampling_stages.py", + "--attention", + "sage2", + "--steps", + "1", + "--warmup-runs", + "1", + "--uninstrumented", + "--cuda-profiler-capture", + "--output", + "/output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-warmed-step-capture-20260826.json" + ], + "environment_switches": { + "CUDA_DEVICE_MAX_CONNECTIONS": "1", + "CUDA_DEVICE_MAX_COPY_CONNECTIONS": "4", + "CUDA_HOME": "/usr/local/cuda", + "CUDA_INC_PATH": "/usr/local/cuda/include", + "CUDA_INJECTION64_PATH": "/opt/nsys/target-linux-sbsa-armv8/libToolsInjection64.so", + "CUDA_MANAGED_FORCE_DEVICE_ALLOC": "1", + "CUDA_MODULE_LOADING": "EAGER", + "CUDA_VERSION": "13.0.2", + "H3_FUSED_ELEMENTWISE": "1", + "H3_MODEL_PATH": "/models/minimax_h3_ref2va_pruned_nvfp4.safetensors", + "H3_NVFP4_FC2_LT_SPLITK1": "1", + "H3_NVFP4_MODULATE_FUSION": "1", + "H3_NVFP4_SCALE_BACKEND": "vortex", + "H3_NVFP4_SCALE_VERSION": "1", + "H3_NVFP4_SWIGLU_FUSION": "1", + "H3_SAGE_QKV_LAYOUT": "strided_nhd", + "TORCH_COMPILE_DISABLE": "0", + "TORCH_CUDA_ARCH_LIST": "12.1a", + "TORCH_EXTENSIONS_DIR": "/opt/h3-blackwell-runtime/.torch_extensions" + }, + "elapsed_seconds": 20.90812512299999, + "stage_trace": [], + "checksums": [ + -197605.3125, + 528.5980224609375 + ], + "sha256": { + "video": "b5fc9bd43ff65f189797fd2377d5a48d8d57f64fb2774fb729dc2be478c39a4b", + "audio": "de6da2540b639a821be7d069efcbf067b209b20de3f7daa0bb9ee28e2624764c" + }, + "fc2_dispatch_delta": { + "attempts": 50, + "successes": 50, + "fallbacks": 0 + }, + "peak_allocated_bytes": 15570401280, + "peak_reserved_bytes": 18538823680, + "measurement_policy": "uninstrumented sampling wall time" +} diff --git a/benchmarks/gb10-post-fc2-warmed-step-nsys-summary-20260826.json b/benchmarks/gb10-post-fc2-warmed-step-nsys-summary-20260826.json new file mode 100644 index 0000000..d0076a0 --- /dev/null +++ b/benchmarks/gb10-post-fc2-warmed-step-nsys-summary-20260826.json @@ -0,0 +1,61 @@ +{ + "source_gpu_trace": "benchmarks\\gb10-post-fc2-warmed-step-stats_cuda_gpu_trace.csv", + "source_kernel_exec_trace": "benchmarks\\gb10-post-fc2-warmed-step-stats_cuda_kern_exec_trace.csv", + "gpu_span_ns": 20907172640, + "gpu_operation_count": 2803, + "kernel_count": 2744, + "kernel_time_ns": 20895065376, + "kernel_busy_percent_of_span": 99.94209038109325, + "launch_gaps": { + "positive_gap_count": 2743, + "total_ns": 11518912, + "average_ns": 4199.384615384615, + "maximum_ns": 4797728 + }, + "cpu_gpu_overlap": { + "scope": "CUDA kernel-launch API intervals intersected with GPU kernel intervals", + "launch_api_union_ns": 13065552528, + "launch_api_gpu_overlap_ns": 13062771136, + "launch_api_overlap_percent": 99.97871202160002 + }, + "components": { + "sage2": { + "milliseconds": 13029.23856, + "percent_of_kernel_time": 62.355576905566124, + "launches": 250 + }, + "nvfp4_gemms": { + "milliseconds": 4167.855808, + "percent_of_kernel_time": 19.946603339117498, + "launches": 200 + }, + "nvfp4_packing": { + "milliseconds": 2026.853664, + "percent_of_kernel_time": 9.700154689767265, + "launches": 800 + }, + "norm_and_rope": { + "milliseconds": 1084.3856, + "percent_of_kernel_time": 5.1896731619970025, + "launches": 202 + }, + "remaining_gate_add": { + "milliseconds": 549.298592, + "percent_of_kernel_time": 2.6288436150619683, + "launches": 100 + }, + "other": { + "milliseconds": 37.433152, + "percent_of_kernel_time": 0.17914828849014078, + "launches": 1192 + } + }, + "classification_policy": { + "sage2": "mainloop plus MeanScale/TransposePadPermute/QuantInt8 preparation", + "nvfp4_gemms": "SM120 block-scaled CUTLASS GEMMs", + "nvfp4_packing": "absmax, final-scale, zero-fill and NVFP4 quantization kernels", + "norm_and_rope": "layer norm, mean reduction and fused RMS/RoPE kernels", + "remaining_gate_add": "fused residual gate/add kernels", + "other": "all unmatched kernels" + } +} diff --git a/benchmarks/gb10-post-fc2-warmed-step-stats_cuda_api_sum.csv b/benchmarks/gb10-post-fc2-warmed-step-stats_cuda_api_sum.csv new file mode 100644 index 0000000..dd03538 --- /dev/null +++ b/benchmarks/gb10-post-fc2-warmed-step-stats_cuda_api_sum.csv @@ -0,0 +1,11 @@ +Time (%),Total Time (ns),Num Calls,Avg (ns),Med (ns),Min (ns),Max (ns),StdDev (ns),Name +60.5,12620672144,2441,5170287.6,4016.0,2384,243250816,27751294.3,cudaLaunchKernel +36.5,7617076416,7,1088153773.7,7616.0,1712,7607043680,2874562566.0,cudaStreamSynchronize +2.1,441237232,300,1470790.8,3744.0,2448,10733504,3261468.3,cuLaunchKernelEx +0.8,163999584,52,3153838.2,5079656.0,128,5558272,2619562.7,cudaMemsetAsync +0.0,3643152,3,1214384.0,5552.0,5456,3632144,2093841.6,cuLaunchKernel +0.0,922240,1,922240.0,922240.0,922240,922240,0.0,cuProfilerStart +0.0,280112,9,31123.6,31344.0,2848,58160,22382.3,cudaMemcpyAsync +0.0,204368,2441,83.7,48.0,16,6592,205.4,cuKernelGetName +0.0,10128,3,3376.0,2016.0,1648,6464,2680.6,cuKernelGetFunction +0.0,5456,1,5456.0,5456.0,5456,5456,0.0,cudaDeviceSynchronize diff --git a/benchmarks/gb10-post-fc2-warmed-step-stats_cuda_gpu_kern_sum.csv b/benchmarks/gb10-post-fc2-warmed-step-stats_cuda_gpu_kern_sum.csv new file mode 100644 index 0000000..76e4d91 --- /dev/null +++ b/benchmarks/gb10-post-fc2-warmed-step-stats_cuda_gpu_kern_sum.csv @@ -0,0 +1,63 @@ +Time (%),Total Time (ns),Instances,Avg (ns),Med (ns),Min (ns),Max (ns),StdDev (ns),Name +57.9,12095370432,50,241907408.6,241960816.0,238904576,245704832,1430443.8,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +19.9,4167855808,200,20839279.0,20371696.0,7702240,35071936,9954629.7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +3.0,616933984,50,12338679.7,12314672.0,11865984,12928448,384444.2,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +2.9,606392320,50,12127846.4,12134352.0,11797920,12481600,272664.4,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +2.6,549298592,100,5492985.9,5476528.0,5449216,5824160,55130.6,_gate_add_kernel +2.5,528098848,50,10561977.0,10468928.0,10341376,11079968,201841.8,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +1.8,371120608,100,3711206.1,3518080.0,3408320,4344608,319034.1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +1.7,358825984,102,3517901.8,3507296.0,45056,3844544,361846.1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +1.4,286555520,50,5731110.4,5711552.0,5660672,5912992,64887.3,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +1.3,264825568,50,5296511.4,5240688.0,5065568,5700544,172507.0,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +1.2,248751328,100,2487513.3,2438752.0,2417696,2845728,108528.2,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +0.9,192832480,50,3856649.6,3831584.0,3724512,4098912,116494.0,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +0.9,189654560,50,3793091.2,3766976.0,3707360,4001984,83318.4,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +0.7,138351232,50,2767024.6,2761616.0,2747456,2899360,25808.7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +0.6,119167296,50,2383345.9,2358464.0,2338976,2551328,51731.7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +0.5,108606272,50,2172125.4,2172144.0,2156896,2190848,8057.6,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +0.1,14493152,200,72465.8,48272.0,40608,206816,42011.8,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +0.0,6883840,2,3441920.0,3441920.0,61632,6822208,4780449.1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +0.0,5532736,60,92212.3,2912.0,1280,5035136,649588.6,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +0.0,5508544,2,2754272.0,2754272.0,51296,5457248,3822585.3,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +0.0,4915936,1,4915936.0,4915936.0,4915936,4915936,0.0,void cutlass::Kernel2(T1::Params) +0.0,4410976,1,4410976.0,4410976.0,4410976,4410976,0.0,"void magma_sgemmEx_kernel(int, int, int, Tensor, int, Tensor, int, Tensor, int, Tensor, int, int, int, const T1 *, const T1 *, T1, T1, int, cublasLtEpilogue_t, int, const void *, long)" +0.0,3766816,1,3766816.0,3766816.0,3766816,3766816,0.0,"void at::native::::CatArrayBatchedCopy_alignedK_contig::OpaqueType<(unsigned int)2>, unsigned int, (int)2, (int)128, (int)1, (int)8>(T1 *, at::native::::CatArrInputTensorMetadata, at::native::::TensorSizeStride, int, T2)" +0.0,1542400,51,30243.1,26560.0,4640,93376,13266.2,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +0.0,873696,1,873696.0,873696.0,873696,873696,0.0,"void at::native::::CatArrayBatchedCopy_alignedK_contig::OpaqueType<(unsigned int)4>, unsigned int, (int)3, (int)128, (int)1, (int)16>(T1 *, at::native::::CatArrInputTensorMetadata, at::native::::TensorSizeStride, int, T2)" +0.0,730688,201,3635.3,2336.0,1920,100448,7972.3,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +0.0,501280,6,83546.7,81136.0,1600,172960,89454.6,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add, std::array>(int, T2, T3)" +0.0,405472,200,2027.4,1920.0,1824,7584,687.4,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +0.0,400288,150,2668.6,2176.0,1920,9696,1031.2,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +0.0,358944,102,3519.1,4304.0,1600,11488,2037.3,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +0.0,224736,55,4086.1,1536.0,1504,71520,12382.0,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +0.0,218368,5,43673.6,27680.0,2240,88064,35553.0,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +0.0,217312,51,4261.0,3456.0,3040,22080,3140.9,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +0.0,126112,2,63056.0,63056.0,39712,86400,33013.4,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::neg_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +0.0,114816,51,2251.3,1504.0,1472,14048,2514.1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +0.0,108896,52,2094.2,1952.0,1920,3904,385.3,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +0.0,104288,51,2044.9,1312.0,1280,3744,1021.8,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +0.0,99968,1,99968.0,99968.0,99968,99968,0.0,void cutlass::Kernel2(T1::Params) +0.0,97952,50,1959.0,1952.0,1920,2048,29.1,"::final_scale_warp_kernel(const float *, float *, long, float)" +0.0,89600,50,1792.0,896.0,832,8672,1561.9,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +0.0,87776,2,43888.0,43888.0,3904,83872,56545.9,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +0.0,87744,60,1462.4,1440.0,1088,3104,258.8,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +0.0,83392,51,1635.1,1280.0,1216,8192,1333.7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +0.0,78720,51,1543.5,1344.0,1312,6144,751.5,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +0.0,73152,51,1434.4,1120.0,1088,7584,1226.3,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +0.0,52384,1,52384.0,52384.0,52384,52384,0.0,"void at::native::::CatArrayBatchedCopy::OpaqueType<(unsigned int)4>, unsigned int, (int)2, (int)64, (int)64>(T1 *, at::native::::CatArrInputTensorMetadata, at::native::::TensorSizeStride, int, T2)" +0.0,48832,1,48832.0,48832.0,48832,48832,0.0,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +0.0,44096,1,44096.0,44096.0,44096,44096,0.0,void cutlass::Kernel2(T1::Params) +0.0,39456,1,39456.0,39456.0,39456,39456,0.0,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::sin_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +0.0,31872,1,31872.0,31872.0,31872,31872,0.0,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::cos_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +0.0,16448,7,2349.7,1376.0,1056,6688,2004.8,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +0.0,11424,5,2284.8,2080.0,1824,3392,628.1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +0.0,8960,2,4480.0,4480.0,3360,5600,1583.9,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +0.0,6112,2,3056.0,3056.0,1184,4928,2647.4,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithoutCast, at::native::memory::StoreWithoutCast>(int, T1, T2, T4, T5, T6, T7)" +0.0,5728,1,5728.0,5728.0,5728,5728,0.0,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +0.0,4672,2,2336.0,2336.0,1824,2848,724.1,"void at::native::unrolled_elementwise_kernel::CompareEqFunctor>, std::array, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithoutCast, at::native::memory::StoreWithoutCast>(int, T1, T2, T4, T5, T6, T7)" +0.0,4128,1,4128.0,4128.0,4128,4128,0.0,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl_nocast>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +0.0,4000,1,4000.0,4000.0,4000,4000,0.0,"void cublasLt::splitKreduce_kernel<(int)32, (int)16, int, float, float, float, float, (bool)0, float, float, float, (bool)1, (bool)1, (bool)0, (bool)0>(cublasLt::cublasSplitKParams, const T4 *, const T10 *, T9 *, T5 *, const T6 *, const T6 *, const T11 *, const T4 *, T11 *, void *, long, T6 *, int *, T6 *, T6 *, const T6 *, const T6 *, const T6 *, const T6 *, const T6 *)" +0.0,3872,2,1936.0,1936.0,1024,2848,1289.8,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnOther_add, std::array>(int, T2, T3)" +0.0,1984,1,1984.0,1984.0,1984,1984,0.0,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BUnaryFunctor>, std::array>(int, T2, T3)" +0.0,1664,1,1664.0,1664.0,1664,1664,0.0,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::neg_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 9)]::operator ()() const::[lambda(c10::BFloat16) (instance 1)], std::array>(int, T2, T3)" +0.0,1312,1,1312.0,1312.0,1312,1312,0.0,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" diff --git a/benchmarks/gb10-post-fc2-warmed-step-stats_cuda_gpu_trace.csv b/benchmarks/gb10-post-fc2-warmed-step-stats_cuda_gpu_trace.csv new file mode 100644 index 0000000..06444f1 --- /dev/null +++ b/benchmarks/gb10-post-fc2-warmed-step-stats_cuda_gpu_trace.csv @@ -0,0 +1,2804 @@ +Start (ns),Duration (ns),CorrId,GrdX,GrdY,GrdZ,BlkX,BlkY,BlkZ,Reg/Trd,StcSMem (MB),DymSMem (MB),Bytes (MB),Throughput (MB/s),SrcMemKd,DstMemKd,Device,Ctx,GreenCtx,Strm,Name +6687024,129408,4655,,,,,,,,,,14.322,110663.498,Device,Device,NVIDIA GB10 (0),1,,7,[CUDA memcpy Device-to-Device] +6818800,4288,4667,,,,,,,,,,0.053,12358.158,Device,Device,NVIDIA GB10 (0),1,,7,[CUDA memcpy Device-to-Device] +7204912,1888,4680,,,,,,,,,,0.000,4.237,Pageable,Device,NVIDIA GB10 (0),1,,7,[CUDA memcpy Host-to-Device] +7275376,2144,4697,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +7292688,2816,4708,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +7305872,3392,4719,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7313232,1088,4730,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +7320496,1088,4741,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +7326704,1056,4752,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +7332624,2080,4763,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7339504,1824,4774,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7357872,2752,4788,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +7370736,5728,4800,26,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7381104,1120,4811,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +7392976,1984,4822,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BUnaryFunctor>, std::array>(int, T2, T3)" +7417616,38432,4839,3497,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +7459632,48832,4854,6993,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7510000,71520,4869,6993,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +7647600,4915936,4896,2336,3,1,256,1,1,212,0.000,0.049,,,,,NVIDIA GB10 (0),1,,7,void cutlass::Kernel2(T1::Params) +12565488,5035136,4909,195804,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +17601648,3808,4927,26,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +17607760,44096,4953,56,6,1,128,1,1,130,0.000,0.031,,,,,NVIDIA GB10 (0),1,,7,void cutlass::Kernel2(T1::Params) +17653744,30368,4966,2174,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +17685488,3744,4978,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +17691728,1792,4989,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +17696080,1344,5000,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +17700080,2048,5011,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +17704272,1536,5022,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +17708368,1280,5033,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +17712368,1376,5044,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +17716560,2080,5055,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +17720400,2848,5066,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnOther_add, std::array>(int, T2, T3)" +17729712,6688,5072,,,,,,,,,,0.000,0.598,Device,Pinned,NVIDIA GB10 (0),1,,7,[CUDA memcpy Device-to-Host] +17749968,1024,5083,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnOther_add, std::array>(int, T2, T3)" +17754544,608,5089,,,,,,,,,,0.000,6.579,Device,Pinned,NVIDIA GB10 (0),1,,7,[CUDA memcpy Device-to-Host] +17778576,960,5102,,,,,,,,,,0.000,4.167,Pageable,Device,NVIDIA GB10 (0),1,,7,[CUDA memcpy Host-to-Device] +17807664,3766816,5114,1536,3,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::CatArrayBatchedCopy_alignedK_contig::OpaqueType<(unsigned int)2>, unsigned int, (int)2, (int)128, (int)1, (int)8>(T1 *, at::native::::CatArrInputTensorMetadata, at::native::::TensorSizeStride, int, T2)" +26302704,20288,5127,,,,,,,,,,0.907,44727.718,Pageable,Device,NVIDIA GB10 (0),1,,7,[CUDA memcpy Host-to-Device] +26372208,8192,5141,222,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +26413968,27680,5153,7090,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +26449776,52384,5165,96,3,1,512,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::CatArrayBatchedCopy::OpaqueType<(unsigned int)4>, unsigned int, (int)2, (int)64, (int)64>(T1 *, at::native::::CatArrInputTensorMetadata, at::native::::TensorSizeStride, int, T2)" +26503152,31872,5176,1773,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::cos_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +26536016,39456,5187,1773,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::sin_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +26577904,39712,5198,1773,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::neg_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +26618864,873696,5210,1536,4,1,128,1,1,37,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::CatArrayBatchedCopy_alignedK_contig::OpaqueType<(unsigned int)4>, unsigned int, (int)3, (int)128, (int)1, (int)16>(T1 *, at::native::::CatArrInputTensorMetadata, at::native::::TensorSizeStride, int, T2)" +27495408,217120,5224,7090,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +27714544,2240,5236,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +27718736,1536,5247,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +27722736,3424,5258,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +27728848,3168,5272,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +27733072,3136,5284,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +27737488,4736,5296,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +27743504,1088,5309,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +27747568,1632,5321,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +27751536,1504,5334,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +27755760,2016,5345,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +27759856,30048,5366,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +27792368,3544672,5389,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +31338480,2464,5404,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +31342576,1984,5422,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +31346672,49568,5462,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +31398896,3452992,5470,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +34853840,2912,5473,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +34857968,2466752,5476,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +37326064,1920,5487,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +37330160,24059776,5516,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +61391216,11842304,5533,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +73250640,448,5547,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +73252304,2395136,5550,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +75649264,4001984,5583,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +79653872,3999008,5585,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +83654864,5297792,5601,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +88955088,5731328,5624,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +94689520,240315488,5627,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +335006992,2176224,5634,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +337185008,1952,5637,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +337189104,1376,5651,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +337193200,1504,5666,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +337197296,40608,5689,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +337240272,2775488,5701,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +340017392,1888,5710,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +340021488,7702240,5739,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +347725040,5461056,5753,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +353188080,3717632,5773,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +356907248,5856,5788,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +356915440,3776,5806,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +356921584,60928,5846,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +356984048,4091840,5854,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +361077968,1952,5857,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +361082096,2437024,5860,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +363520432,1920,5871,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +363524336,34563520,5900,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +398089488,140224,5941,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +398231792,11923520,5949,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +410157296,2464,5952,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +410161392,10756480,5955,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +420920560,1920,5966,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +420924656,3040,5981,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +420929776,16202336,6002,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +437133584,5455232,6015,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +442590448,1984,6023,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +442594544,1440,6034,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +442598640,1376,6045,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +442602736,3168,6059,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +442608848,1344,6071,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +442613040,4608,6083,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +442618992,1120,6096,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +442623216,1632,6108,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +442627312,1472,6121,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +442631408,1248,6132,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +442635504,29472,6153,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +442666256,3479584,6176,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +446148848,2368,6191,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +446152944,2112,6209,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +446157040,48416,6249,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +446208240,3449888,6257,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +449660144,4032,6260,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +449666288,2435424,6263,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +452104432,1888,6274,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +452108528,25770432,6303,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +477881584,11819968,6320,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +489703312,416,6334,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +489704624,2346688,6337,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +492053744,3809824,6370,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +495865072,3947776,6372,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +499815664,5406624,6388,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +505224432,5664096,6411,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +510891248,240444576,6414,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +751338768,2172448,6421,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +753512656,1952,6424,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +753516784,2912,6438,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +753520976,1536,6453,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +753525008,46304,6476,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +753573072,2749568,6488,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +756324592,1920,6497,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +756328656,7743936,6526,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +764074288,5461696,6540,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +769538256,3462944,6560,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +773002480,2304,6575,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +773006576,1984,6593,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +773010672,46624,6633,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +773058800,3966080,6641,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +777027824,1952,6644,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +777031920,2729952,6647,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +779763952,1856,6658,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +779768048,33135360,6687,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +812905712,186080,6728,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +813094096,12695904,6736,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +825791728,2176,6739,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +825795920,10475328,6742,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +836272528,2176,6753,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +836276464,832,6768,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +836278352,15614944,6789,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +851894576,5572000,6802,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +857469168,2336,6810,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +857473264,3104,6821,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +857479408,3264,6832,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +857485552,11680,6846,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +857499856,6144,6858,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +857509360,5600,6870,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +857516272,2368,6883,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +857520368,3360,6895,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +857526512,2944,6908,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +857530736,8192,6919,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +857540848,56608,6940,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +857600240,3673856,6963,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +861275376,2464,6978,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +861279440,2048,6996,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +861283536,48960,7036,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +861334736,3813056,7044,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +865150160,3776,7047,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +865156304,2434944,7050,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +867593456,1888,7061,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +867597584,23937632,7090,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +891537648,12480864,7107,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +904019856,416,7121,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +904021456,2345600,7124,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +906369264,3709440,7157,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +910080208,3724512,7159,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +913807568,5534656,7175,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +919343504,5795488,7198,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +925141232,239349632,7201,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +1164493200,2170784,7208,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +1166665936,1920,7211,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +1166670064,2912,7225,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +1166674288,1504,7240,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +1166678288,47104,7263,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +1166728400,2757664,7275,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +1169489136,1952,7284,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +1169493232,7799744,7313,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +1177295120,5451936,7327,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +1182748880,3475680,7347,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +1186227440,2400,7362,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +1186231504,2080,7380,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +1186235600,46016,7420,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +1186282896,3462240,7428,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +1189746928,1984,7431,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +1189750992,2452064,7434,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +1192204528,1888,7445,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +1192208624,34645696,7474,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +1226856784,139968,7515,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +1226999024,12761440,7523,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +1239763184,1920,7526,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +1239767280,10456096,7529,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +1250225392,1952,7540,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +1250229456,864,7555,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +1250231632,15792128,7576,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +1266026768,5824160,7589,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +1271852272,1952,7597,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +1271856368,1408,7608,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +1271860432,2848,7619,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +1271864560,3200,7633,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +1271870704,1376,7645,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +1271874800,4416,7657,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +1271880912,1120,7670,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +1271885040,1600,7682,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +1271889136,1472,7695,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +1271893200,1280,7706,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +1271897392,26016,7727,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +1271925968,3545984,7750,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +1275473232,2496,7765,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +1275477232,2112,7783,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +1275481328,47552,7823,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +1275530768,4259520,7831,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +1279793424,3712,7834,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +1279799536,2444704,7837,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +1282245872,1824,7848,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +1282249936,24882560,7877,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +1307135248,12475168,7894,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +1319611344,416,7908,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +1319612656,2339296,7911,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +1321954544,3722080,7944,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +1325678832,3736032,7946,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +1329416432,5150368,7962,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +1334569200,5873152,7985,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +1340443888,238904576,7988,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +1579351312,2158912,7995,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +1581512912,1984,7998,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +1581517040,2880,8012,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +1581521200,1536,8027,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +1581525136,54208,8050,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +1581580624,2747456,8062,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +1584329936,1920,8071,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +1584334064,7804864,8100,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +1592141072,5612512,8114,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +1597755632,3661824,8134,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +1601419504,2400,8149,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +1601423568,2112,8167,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +1601427696,46240,8207,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +1601475824,3859104,8215,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +1605337328,1952,8218,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +1605341456,2438848,8221,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +1607782640,1920,8232,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +1607786736,34043744,8261,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +1641831824,139840,8302,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +1641972944,12366720,8310,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +1654341872,1920,8313,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +1654345968,10667840,8316,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +1665016048,1984,8327,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +1665020144,2560,8342,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +1665024240,15704512,8363,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +1680730352,5454720,8376,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +1686186480,1952,8384,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +1686189744,1440,8395,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +1686193392,1312,8406,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +1686197488,3264,8420,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +1686203600,1344,8432,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +1686207728,4640,8444,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +1686213872,1088,8457,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +1686217968,1632,8469,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +1686222160,1504,8482,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +1686226160,1280,8493,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +1686230256,27264,8514,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +1686258928,3493056,8537,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +1689753872,2464,8552,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +1689757936,2016,8570,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +1689762032,47456,8610,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +1689811152,3918208,8618,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +1693731056,3392,8621,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +1693737200,2845728,8624,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +1696585968,1856,8635,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +1696590096,25301536,8664,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +1721894096,11901664,8681,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +1733796816,384,8695,,,,,,,,,,0.000,583.333,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +1733798384,2551328,8698,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +1736351984,3738560,8731,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +1740092624,3951712,8733,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +1744046320,5119744,8749,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +1749167344,5697824,8772,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +1754867920,240193088,8775,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +1995062544,2167328,8782,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +1997232368,1920,8785,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +1997236432,2944,8799,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +1997240656,1504,8814,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +1997244624,47296,8837,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +1997293808,2750016,8849,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +2000046288,1952,8858,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +2000050416,7806304,8887,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +2007859472,5472288,8901,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +2013334768,3708736,8921,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +2017045712,6624,8936,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2017053936,2144,8954,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2017058000,47936,8994,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2017107216,4063680,9002,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +2021173488,1952,9005,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +2021177584,2438432,9008,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +2023617776,1920,9019,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +2023621872,34049568,9048,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +2057673968,139872,9089,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2057816272,11888096,9097,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +2069705936,1984,9100,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +2069710064,10741088,9103,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +2080453872,1952,9114,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +2080457968,2496,9129,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2080462064,16236448,9150,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +2096700624,5467584,9163,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +2102170832,1952,9171,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +2102174960,1440,9182,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +2102179056,1312,9193,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +2102181904,3456,9207,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +2102187248,1344,9219,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +2102191280,4608,9231,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +2102197488,1088,9244,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +2102201520,1632,9256,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +2102205680,1728,9269,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +2102209776,1280,9280,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2102213840,26368,9301,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +2102242544,3495232,9324,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +2105740528,2496,9339,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2105744624,1984,9357,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2105748720,47232,9397,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2105797840,3515456,9405,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +2109316304,3360,9408,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +2109322480,2434016,9411,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +2111758576,1888,9422,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +2111762672,25849184,9451,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +2137614608,11818912,9468,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +2149435280,416,9482,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +2149436720,2354816,9485,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +2151792912,3750336,9518,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +2155544784,4064384,9520,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +2159612112,5416480,9536,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +2165031152,5667744,9559,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +2170700176,242080160,9562,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +2412782864,2159488,9569,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +2414944496,1952,9572,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +2414948592,2944,9586,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +2414952560,1504,9601,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +2414955312,49536,9624,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2415006928,2752224,9636,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +2417760496,1888,9645,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +2417764560,7814080,9674,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +2425580816,5456192,9688,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +2431038672,3483520,9708,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +2434524624,6336,9723,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2434532592,36384,9741,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2434571504,54208,9781,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2434628848,4283872,9789,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +2438914256,1952,9792,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +2438918384,2440160,9795,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +2441360624,1888,9806,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +2441364720,33763776,9835,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +2475131248,139488,9876,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2475272432,12292416,9884,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +2487567568,2176,9887,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +2487571728,10786304,9890,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +2498360560,1920,9901,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +2498364656,832,9916,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2498366768,15820672,9937,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +2514190576,5490432,9950,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +2519683344,2272,9958,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +2519687440,1408,9969,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +2519691504,3712,9980,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +2519697648,8064,9994,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +2519707856,1568,10006,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +2519711920,8096,10018,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +2519722224,6880,10031,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +2519730384,1632,10043,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +2519734512,14048,10056,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +2519750896,3648,10067,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2519757040,93376,10088,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +2519853264,3592384,10111,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +2523448560,2912,10126,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2523452720,2240,10144,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2523456752,47712,10184,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2523505904,3622240,10192,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +2527129808,4000,10195,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +2527135984,2437120,10198,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +2529576208,1856,10209,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +2529580272,25405312,10238,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +2554986864,12071296,10255,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +2567060336,448,10269,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +2567062000,2347552,10272,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +2569410832,3711616,10305,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +2573123824,4098912,10307,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +2577223984,5356480,10323,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +2582582512,5691456,10346,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +2588277008,239531264,10349,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +2827810064,2158528,10356,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +2829970640,1984,10359,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +2829974768,2720,10373,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +2829978864,1504,10388,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +2829982928,46080,10411,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2830032080,2749120,10423,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +2832783600,1920,10432,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +2832787696,7819616,10461,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +2840610032,5449728,10475,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +2846062832,3479168,10495,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +2849543408,2336,10510,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2849547504,1984,10528,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2849551600,46464,10568,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2849600720,3413024,10576,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +2853015792,1984,10579,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +2853019888,2757152,10582,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +2855779536,1856,10593,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +2855783632,33791744,10622,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +2889576720,141536,10663,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2889721072,12811136,10671,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +2902534352,2208,10674,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +2902538480,10362432,10677,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +2912903408,1952,10688,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +2912907472,864,10703,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2912909616,15744352,10724,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +2928656624,5465664,10737,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +2934124752,1984,10745,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +2934128912,1408,10756,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +2934132976,3680,10767,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +2934139120,3488,10781,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +2934145136,1344,10793,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +2934147728,4608,10805,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +2934154352,1120,10818,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +2934158576,1632,10830,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +2934162672,1472,10843,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +2934166768,1280,10854,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2934170864,25920,10875,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +2934199536,3495488,10898,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +2937696496,2528,10913,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2937700560,1920,10931,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2937704656,48864,10971,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2937754864,3453632,10979,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +2941210832,3808,10982,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +2941217008,2435392,10985,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +2943654160,1888,10996,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +2943658224,24871360,11025,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +2968532176,12473728,11042,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +2981008304,416,11056,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +2981009616,2342880,11059,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +2983354608,3711520,11092,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +2987067632,3730304,11094,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +2990800112,5451680,11110,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +2996253392,5726816,11133,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +3001983216,240209920,11136,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +3242196240,2180576,11143,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +3244379376,1984,11146,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +3244383504,2752,11160,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +3244387536,1504,11175,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +3244391664,45632,11198,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +3244438736,2899360,11210,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +3247340784,2016,11219,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +3247344848,8260608,11248,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +3255608560,5606272,11262,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +3261217040,3532864,11282,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +3264751824,2528,11297,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +3264755920,1984,11315,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +3264760048,46304,11355,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +3264808144,3517760,11363,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +3268327664,2144,11366,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +3268331760,2417696,11369,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +3270751472,1888,11380,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +3270755568,34255616,11409,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +3305013552,141536,11450,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +3305157872,12819968,11458,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +3317979344,2464,11461,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +3317983472,10346848,11464,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +3328333040,2144,11475,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +3328337136,864,11490,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +3328339280,15735488,11511,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +3344077040,5454016,11524,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +3349532912,2016,11532,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +3349537008,1408,11543,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +3349541104,1312,11554,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +3349545200,3456,11568,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +3349551344,1312,11580,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +3349555568,4224,11592,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +3349561520,1088,11605,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +3349565680,1632,11617,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +3349569744,1504,11630,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +3349573872,1280,11641,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +3349577936,26720,11662,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +3349606640,3496640,11685,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +3353104752,2400,11700,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +3353108720,2112,11718,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +3353112816,47680,11758,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +3353162960,3506560,11766,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +3356672208,3680,11769,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +3356678352,2438752,11772,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +3359119600,1888,11783,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +3359123664,24909184,11812,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +3384035568,12198048,11829,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +3396235152,416,11843,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +3396236784,2509216,11846,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +3398747472,3777248,11879,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +3402526000,3770368,11881,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +3406299344,5082752,11897,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +3411384560,5710464,11920,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +3417097904,240101408,11923,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +3657200912,2166688,11930,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +3659370736,1920,11933,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +3659374832,2528,11947,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +3659378896,1536,11962,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +3659382896,45856,11985,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +3659431120,2762624,11997,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +3662195920,1920,12006,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +3662200016,8099264,12035,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +3670300976,5523968,12049,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +3675826448,3822336,12069,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +3679650064,2528,12084,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +3679654128,2080,12102,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +3679658224,46624,12142,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +3679707344,3806144,12150,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +3683514768,1920,12153,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +3683518704,2444096,12156,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +3685965040,1920,12167,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +3685969136,34983616,12196,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +3720954160,139744,12237,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +3721095408,12296512,12245,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +3733394640,9696,12248,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +3733406960,10774656,12251,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +3744184560,7232,12262,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +3744194928,3328,12277,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +3744200976,15784672,12298,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +3759986992,5478976,12311,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +3765468400,1952,12319,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +3765472496,1440,12330,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +3765476592,1280,12341,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +3765480656,3456,12355,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +3765486800,1312,12367,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +3765490800,4448,12379,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +3765497040,1312,12392,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +3765499728,1632,12404,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +3765503216,1472,12417,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +3765507312,1280,12428,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +3765511408,24768,12449,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +3765538032,3497504,12472,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +3769037040,2464,12487,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +3769041136,2016,12505,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +3769045232,48544,12545,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +3769095376,3424864,12553,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +3772522768,3552,12556,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +3772528912,2510688,12559,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +3775041776,1888,12570,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +3775045872,24845056,12599,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +3799893232,11830912,12616,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +3811725200,416,12630,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +3811726480,2388928,12633,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +3814117616,3942304,12666,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +3818062032,3950304,12668,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +3822014736,5234368,12684,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +3827250416,5660672,12707,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +3832913136,240325056,12710,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +4073240848,2157632,12717,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +4075400400,1952,12720,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +4075404496,2560,12734,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +4075408624,1504,12749,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +4075412784,50400,12772,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +4075465936,2760608,12784,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +4078227856,1952,12793,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +4078231792,7826784,12822,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +4086061328,5463200,12836,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +4091526352,3570432,12856,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +4095099184,2336,12871,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +4095103184,2208,12889,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +4095107312,48384,12929,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +4095157488,4305856,12937,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +4099465456,1920,12940,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +4099469584,2451072,12943,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +4101923088,1856,12954,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +4101927120,33911744,12983,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +4135841008,206816,13024,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +4136050928,12091680,13032,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +4148144336,2208,13035,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +4148148464,10751008,13038,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +4158902512,1920,13049,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +4158906608,3328,13064,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +4158912752,15259232,13085,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +4174173648,5466400,13098,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +4179642672,1920,13106,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +4179646704,1440,13117,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +4179650768,1312,13128,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +4179654896,3072,13142,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +4179661040,1312,13154,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +4179665040,4800,13166,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +4179671248,1120,13179,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +4179675312,1632,13191,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +4179679472,1568,13204,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +4179683536,1248,13215,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +4179687696,28672,13236,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +4179718384,3498816,13259,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +4183218480,2432,13274,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +4183222512,2112,13292,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +4183226576,48096,13332,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +4183276752,3518400,13340,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +4186796464,3456,13343,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +4186802448,2435904,13346,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +4189240560,1888,13357,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +4189244656,26090176,13386,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +4215337200,12096896,13403,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +4227435376,416,13417,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +4227437008,2347104,13420,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +4229786864,3730976,13453,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +4233519312,4067328,13455,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +4237588688,5428096,13471,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +4243019024,5745408,13494,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +4248766704,242399936,13497,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +4491168048,2168800,13504,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +4493338896,1920,13507,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +4493342960,2880,13521,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +4493347184,1504,13536,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +4493351184,46592,13559,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +4493399248,2748320,13571,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +4496149744,1952,13580,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +4496153872,7929856,13609,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +4504086576,5458016,13623,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +4509546736,3492480,13643,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +4513041744,2400,13658,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +4513045712,1984,13676,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +4513049840,47232,13716,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +4513100016,4172256,13724,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +4517273808,1984,13727,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +4517277936,2537024,13730,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +4519816432,1888,13741,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +4519820560,33533088,13770,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +4553355536,157600,13811,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +4553515248,12645408,13819,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +4566162640,2432,13822,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +4566166768,10464160,13825,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +4576632208,2400,13836,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +4576636144,2208,13851,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +4576640240,15692288,13872,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +4592334064,5631424,13885,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +4597968112,2560,13893,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +4597972208,1440,13904,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +4597976272,1472,13915,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +4597980368,3680,13929,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +4597986480,3136,13941,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +4597992688,6432,13953,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +4598000880,1280,13966,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +4598004976,7072,13978,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +4598013424,11168,13991,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +4598027504,1248,14002,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +4598031600,71488,14023,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +4598105296,3607424,14046,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +4601714000,2464,14061,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +4601718000,2080,14079,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +4601722096,48672,14119,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +4601772304,3869632,14127,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +4605645008,4032,14130,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +4605651152,2434688,14133,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +4608087280,1856,14144,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +4608091408,25006880,14173,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +4633100528,12395104,14190,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +4645497744,416,14204,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +4645499056,2352544,14207,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +4647854320,3720192,14240,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +4651577584,3872096,14242,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +4655452496,5580512,14258,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +4661034288,5712640,14281,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +4666749168,241404896,14284,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +4908156176,2166912,14291,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +4910324976,2016,14294,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +4910329072,2912,14308,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +4910333264,1536,14323,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +4910337264,48064,14346,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +4910388432,2752704,14358,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +4913142992,1920,14367,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +4913147120,7930720,14396,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +4921079152,5452960,14410,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +4926534896,3469120,14430,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +4930006352,2464,14445,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +4930010320,2016,14463,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +4930014448,46528,14503,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +4930063600,3409568,14511,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +4933475536,1952,14514,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +4933479664,2424224,14517,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +4935905520,1888,14528,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +4935909648,33606624,14557,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +4969519376,139712,14598,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +4969661680,12773024,14606,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +4982436080,2176,14609,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +4982440176,10351520,14612,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +4992792976,1920,14623,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +4992796912,832,14638,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +4992799056,15773088,14659,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +5008574736,5499680,14672,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +5014076656,2496,14680,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +5014080752,1440,14691,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +5014084848,3712,14702,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +5014090992,3584,14716,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +5014097008,1344,14728,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +5014101392,4544,14740,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +5014107408,1088,14753,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +5014111472,1600,14765,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +5014115536,1920,14778,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +5014119664,1248,14789,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +5014123728,26144,14810,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +5014152496,3804640,14833,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +5017958640,22080,14848,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +5017983184,32192,14866,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +5018018032,47840,14906,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +5018068176,3836736,14914,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +5021906256,3744,14917,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +5021912336,2440928,14920,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +5024354576,1856,14931,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +5024358640,24941184,14960,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +5049301200,12450592,14977,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +5061753712,448,14991,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +5061755344,2347648,14994,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +5064105200,3717760,15027,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +5067824464,3725984,15029,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +5071551696,5502592,15045,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +5077055728,5676000,15068,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +5082734864,240707424,15071,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +5323444496,2169536,15078,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +5325615312,1952,15081,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +5325619440,2880,15095,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +5325623600,1536,15110,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +5325627632,47104,15133,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +5325676752,2767680,15145,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +5328446704,1920,15154,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +5328450800,8481312,15183,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +5336933648,5526464,15197,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +5342462192,3485760,15217,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +5345949936,2400,15232,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +5345954032,1984,15250,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +5345958128,46688,15290,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +5346006224,3422112,15298,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +5349430640,1952,15301,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +5349434576,2419200,15304,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +5351855344,1888,15315,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +5351859408,34284960,15344,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +5386147088,139712,15385,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +5386289392,12727072,15393,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +5399018704,1984,15396,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +5399022864,10358368,15399,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +5409383696,1920,15410,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +5409387760,832,15425,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +5409389872,15717632,15446,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +5425110256,5460384,15459,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +5430572240,1952,15467,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +5430576368,1440,15478,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +5430580464,1312,15489,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +5430584560,3392,15503,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +5430590672,1376,15515,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +5430594800,4256,15527,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +5430600976,1088,15540,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +5430605040,1600,15552,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +5430609104,1504,15565,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +5430613328,1248,15576,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +5430617296,25312,15597,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +5430643952,3500224,15620,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +5434146064,2496,15635,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +5434150192,2112,15653,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +5434154224,49952,15693,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +5434206448,4299072,15701,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +5438508272,3648,15704,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +5438514416,2441600,15707,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +5440958704,1888,15718,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +5440962800,24888224,15747,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +5465852336,12351616,15764,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +5478208304,4000,15778,,,,,,,,,,0.000,56.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +5478213936,2420480,15781,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +5480636656,3707936,15814,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +5484346576,3735616,15816,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +5488084176,5065568,15832,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +5493151984,5827808,15855,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +5498981616,241799776,15858,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +5740782896,2166496,15865,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +5742950864,1984,15868,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +5742954704,2912,15882,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +5742958928,1536,15897,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +5742962896,46112,15920,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +5743012048,2754048,15932,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +5745768688,1920,15941,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +5745772752,7818272,15970,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +5753593104,5449504,15984,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +5759044816,3466176,16004,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +5762512304,2304,16019,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +5762516208,1952,16037,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +5762520304,47168,16077,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +5762569456,3411264,16085,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +5765983440,1952,16088,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +5765987568,2425440,16091,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +5768415504,1888,16102,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +5768419568,34226240,16131,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +5802648976,140032,16172,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +5802791152,11884448,16180,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +5814676880,1920,16183,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +5814680848,10341376,16186,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +5825025264,1984,16197,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +5825029360,2560,16212,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +5825033456,15733664,16233,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +5840768400,5464736,16246,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +5846234416,1920,16254,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +5846238480,1408,16265,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +5846242512,1312,16276,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +5846246608,3072,16290,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +5846252752,1344,16302,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +5846256880,4864,16314,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +5846263056,1088,16327,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +5846267152,1664,16339,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +5846271184,1504,16352,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +5846275312,1280,16363,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +5846279376,25952,16384,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +5846308080,3495520,16407,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +5849806032,2368,16422,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +5849810128,2112,16440,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +5849814224,48096,16480,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +5849864432,3438592,16488,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +5853306064,3424,16491,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +5853312368,2821952,16494,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +5856136464,1856,16505,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +5856140528,25366272,16534,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +5881509168,11847616,16551,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +5893358672,416,16565,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +5893360240,2496864,16568,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +5895859440,3826176,16601,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +5899688272,3904192,16603,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +5903594704,5218816,16619,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +5908815088,5683744,16642,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +5914500336,241798560,16645,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +6156301584,2171424,16652,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +6158475504,1984,16655,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +6158479600,2912,16669,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +6158483792,1536,16684,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +6158487792,48224,16707,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +6158538960,2755936,16719,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +6161297648,1952,16728,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +6161301712,7908928,16757,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +6169213200,5461504,16771,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +6174677200,3478752,16791,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +6178157808,2304,16806,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +6178161904,2016,16824,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +6178166000,45856,16864,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +6178214128,3426912,16872,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +6181643472,1984,16875,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +6181647600,2425184,16878,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +6184075504,1888,16889,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +6184079600,34891392,16918,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +6218972432,141280,16959,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +6219116784,11920992,16967,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +6231040240,1952,16970,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +6231044336,10766528,16973,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +6241813744,2144,16984,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +6241817840,832,16999,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +6241819952,16326752,17020,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +6258148592,5467264,17033,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +6263617776,1984,17041,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +6263621904,1408,17052,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +6263625968,1312,17063,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +6263630064,3264,17077,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +6263636208,1440,17089,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +6263640304,4704,17101,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +6263646480,1088,17114,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +6263650544,1600,17126,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +6263654640,1472,17139,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +6263658736,1280,17150,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +6263662832,26688,17171,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +6263691504,3502624,17194,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +6267195632,2400,17209,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +6267199696,1984,17227,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +6267203568,47584,17267,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +6267254000,3522112,17275,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +6270778608,3968,17278,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +6270784752,2439168,17281,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +6273226992,1888,17292,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +6273231088,25829856,17321,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +6299062512,11811904,17338,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +6310876048,416,17352,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +6310877328,2378688,17355,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +6313257552,3908832,17388,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +6317167952,3972352,17390,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +6321143472,5239040,17406,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +6326385040,5665728,17429,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +6332052720,240203808,17432,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +6572259600,2177760,17439,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +6574439664,1952,17442,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +6574443728,2976,17456,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +6574447984,1504,17471,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +6574451920,48864,17494,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +6574502192,2758368,17506,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +6577262832,1920,17515,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +6577266928,7814496,17544,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +6585084176,5459264,17558,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +6590546160,3478464,17578,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +6594025904,2400,17593,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +6594029808,2112,17611,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +6594033904,47328,17651,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +6594083024,3426688,17659,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +6597512432,1952,17662,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +6597516528,2426080,17665,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +6599944464,1856,17676,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +6599948528,33846624,17705,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +6633797872,172064,17746,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +6633972944,12456320,17754,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +6646431952,2208,17757,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +6646436080,10598688,17760,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +6657036656,1952,17771,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +6657040656,864,17786,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +6657042832,15399488,17807,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +6672444656,5615488,17820,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +6678061456,2016,17828,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +6678065456,1664,17839,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +6678069744,1312,17850,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +6678073584,3648,17864,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +6678079696,1344,17876,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +6678083824,11104,17888,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +6678096208,1856,17901,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +6678100208,2080,17913,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +6678104304,4032,17926,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +6678110448,1504,17937,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +6678114544,62208,17958,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +6678178992,3617312,17981,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +6681798928,2560,17996,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +6681803024,2080,18014,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +6681807056,48160,18054,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +6681857264,3838528,18062,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +6685698288,4000,18065,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +6685704432,2440192,18068,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +6688146672,1920,18079,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +6688150768,26015712,18108,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +6714169584,12030144,18125,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +6726202224,544,18139,,,,,,,,,,0.000,411.765,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +6726203984,2354880,18142,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +6728560880,3711456,18175,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +6732274928,3963872,18177,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +6736241872,5437760,18193,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +6741681360,5725024,18216,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +6747408624,239633120,18219,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +6987044112,2163136,18226,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +6989208784,1952,18229,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +6989212944,2688,18243,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +6989217040,1504,18258,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +6989221104,46848,18281,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +6989270352,2756896,18293,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +6992028880,1952,18302,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +6992033040,7841664,18331,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +6999876048,5460608,18345,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +7005338864,3489408,18365,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +7008830704,2496,18380,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7008834768,1984,18398,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7008838864,46144,18438,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7008886992,3410336,18446,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +7012299984,1952,18449,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +7012304112,2667616,18452,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +7014974448,3712,18463,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7014980848,34079488,18492,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +7049062704,139456,18533,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7049203952,12768992,18541,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +7061975248,1984,18544,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +7061979376,10367616,18547,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +7072348496,1920,18558,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7072352496,864,18573,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7072354640,15705376,18594,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +7088061680,5511744,18607,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +7093574960,2272,18615,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +7093578992,1408,18626,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +7093583088,3680,18637,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +7093589200,3616,18651,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +7093595248,1408,18663,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +7093599472,4448,18675,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +7093605616,1088,18688,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +7093609744,1664,18700,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +7093613808,1472,18713,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +7093617936,1280,18724,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7093622000,26848,18745,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +7093650640,3741248,18768,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +7097394416,13184,18783,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7097409136,13472,18801,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7097425136,97696,18841,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7097525488,3970560,18849,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +7101498576,3552,18852,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +7101504752,2437920,18855,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +7103944944,1888,18866,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7103949040,24860704,18895,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +7128811728,12400832,18912,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +7141214064,416,18926,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +7141215376,2338976,18929,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +7143557328,3707360,18962,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +7147267280,3739648,18964,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +7151008976,5512160,18980,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +7156522416,5684192,19003,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +7162209552,242227488,19006,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +7404438896,2180384,19013,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +7406620880,1952,19016,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +7406624976,2560,19030,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +7406629072,1536,19045,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +7406633200,45024,19068,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7406680304,2751680,19080,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +7409433840,1856,19089,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7409437936,7852832,19118,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +7417292080,5460768,19132,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +7422755056,3467584,19152,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +7426225392,2368,19167,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7426229488,2112,19185,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7426233616,45408,19225,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7426281712,3408320,19233,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +7429691600,1952,19236,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +7429695728,2422752,19239,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +7432121584,1856,19250,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7432125648,34878368,19279,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +7467005072,142560,19320,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7467150576,11880896,19328,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +7479033072,1952,19331,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +7479037168,10359680,19334,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +7489399088,6624,19345,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7489407216,1568,19360,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7489411312,15720800,19381,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +7505134864,5521792,19394,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +7510659312,3904,19402,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +7510665424,1440,19413,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +7510669552,3008,19424,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +7510675696,3520,19438,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +7510681776,2048,19450,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +7510685936,6688,19462,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +7510694128,2848,19475,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +7510698288,1600,19487,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +7510702320,2304,19500,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +7510706416,2240,19511,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7510710512,29504,19532,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +7510741296,3520864,19555,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +7514265104,5600,19570,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7514271984,1952,19588,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7514275952,52032,19628,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7514330416,4260224,19636,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +7518592208,1984,19639,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +7518596336,2447712,19642,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +7521046800,1856,19653,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7521050608,25226208,19682,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +7546278096,12453792,19699,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +7558733712,416,19713,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +7558735472,2351712,19716,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +7561090288,3714304,19749,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +7564807376,3786336,19751,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +7568596176,5154144,19767,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +7573753072,5813440,19790,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +7579568368,240922976,19793,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +7820493072,2156896,19800,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +7822651664,1952,19803,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +7822655728,3360,19817,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +7822661840,1568,19832,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +7822665872,51264,19855,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7822719184,2771712,19867,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +7825492144,1952,19876,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7825496304,7849312,19905,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +7833347344,5451136,19919,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +7838800240,3479136,19939,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +7842280656,2368,19954,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7842284752,2144,19972,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7842288880,47424,20012,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7842339056,3417600,20020,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +7845758160,2016,20023,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +7845762288,2423840,20026,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +7848189168,1888,20037,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7848193008,34270688,20066,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +7882465552,140768,20107,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7882607856,11865984,20115,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +7894475120,2464,20118,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +7894479088,10348864,20121,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +7904830704,1952,20132,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7904834800,2656,20147,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7904838896,15733856,20168,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +7920575728,5463552,20181,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +7926041840,1920,20189,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +7926045904,1440,20200,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +7926050000,1312,20211,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +7926054096,3520,20225,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +7926060272,1344,20237,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +7926064368,4992,20249,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +7926070640,1120,20262,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +7926074640,1632,20274,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +7926078704,1504,20287,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +7926082800,1216,20298,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7926086896,25632,20319,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +7926115568,3505824,20342,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +7929623792,2432,20357,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7929627888,1920,20375,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7929631984,47648,20415,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7929681136,3448928,20423,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +7933133040,4032,20426,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +7933139248,2809952,20429,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +7935952112,1856,20440,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7935956272,25293632,20469,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +7961252080,11798880,20486,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +7973053296,448,20500,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +7973054672,2382432,20503,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +7975439600,3919904,20536,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +7979362544,3922944,20538,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +7983287504,5215584,20554,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +7988505808,5690784,20577,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +7994200656,241148000,20580,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +8235351312,2184512,20587,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +8237538512,1984,20590,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +8237542640,3040,20604,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +8237548752,1568,20619,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +8237552784,52064,20642,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +8237606128,2756640,20654,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +8240365808,1888,20663,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +8240369904,7952640,20692,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +8248324368,5449216,20706,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +8253775088,3472800,20726,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +8257249520,2400,20741,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +8257253616,1984,20759,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +8257257712,45696,20799,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +8257305840,3431232,20807,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +8260739280,1952,20810,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +8260743472,2425536,20813,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +8263171312,1888,20824,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +8263175408,34385728,20853,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +8297562448,140160,20894,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +8297705712,11916032,20902,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +8309624048,2432,20905,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +8309628144,10800032,20908,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +8320430320,1920,20919,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +8320434416,832,20934,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +8320436624,16112576,20955,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +8336551152,5465952,20968,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +8342019312,1952,20976,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +8342023440,1440,20987,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +8342027536,1312,20998,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +8342031600,3680,21012,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +8342037776,1312,21024,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +8342041808,4896,21036,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +8342047984,1088,21049,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +8342052080,1632,21061,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +8342056176,1472,21074,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +8342060272,1248,21085,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +8342064368,26112,21106,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +8342093040,3493632,21129,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +8345587952,2560,21144,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +8345592016,1984,21162,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +8345596144,47840,21202,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +8345646320,3444576,21210,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +8349092208,4096,21213,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +8349098224,2448192,21216,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +8351548656,1888,21227,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +8351552784,25738272,21256,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +8377295632,11891808,21273,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +8389188464,544,21287,,,,,,,,,,0.000,411.765,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +8389190192,2346944,21290,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +8391538928,3761664,21323,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +8395303152,4026528,21325,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +8399331536,5417472,21341,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +8404751600,5663296,21364,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +8410416368,241593728,21367,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +8652011792,2164960,21374,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +8654178544,1920,21377,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +8654182640,2976,21391,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +8654186896,1536,21406,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +8654190832,48416,21429,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +8654242032,2752160,21441,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +8656995568,1888,21450,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +8656999664,7942144,21479,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +8664943920,5458208,21493,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +8670404848,3475616,21513,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +8673883376,4320,21528,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +8673889680,19776,21546,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +8673912016,52288,21586,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +8673967312,4322208,21594,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +8678291696,1952,21597,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +8678295760,2444640,21600,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +8680743152,1920,21611,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +8680747248,34577664,21640,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +8715326704,141088,21681,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +8715469072,12382976,21689,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +8727853392,1984,21692,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +8727857360,10443584,21695,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +8738303216,1952,21706,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +8738307312,2592,21721,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +8738311408,15308832,21742,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +8753622256,5479104,21755,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +8759102704,1952,21763,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +8759106800,1408,21774,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +8759110896,3648,21785,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +8759117040,3072,21799,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +8759123184,1376,21811,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +8759127184,4640,21823,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +8759133456,1152,21836,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +8759137264,1600,21848,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +8759141584,1472,21861,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +8759145712,1248,21872,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +8759149776,25216,21893,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +8759176432,3694624,21916,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +8762874096,2400,21931,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +8762878256,2176,21949,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +8762882288,50528,21989,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +8762934480,3638240,21997,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +8766574800,1952,22000,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +8766578928,2442912,22003,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +8769024368,1920,22014,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +8769028368,25552160,22043,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +8794582288,12077248,22060,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +8806660976,448,22074,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +8806662096,2352160,22077,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +8809015568,3721376,22110,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +8812738768,4098432,22112,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +8816839888,5346304,22128,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +8822189264,5725184,22151,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +8827916528,241606016,22154,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +9069525264,2174688,22161,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +9071702320,1952,22164,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +9071706384,2720,22178,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +9071710416,1536,22193,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +9071714512,47232,22216,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +9071763664,2758720,22228,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +9074524400,1952,22237,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +9074528496,7921504,22266,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +9082451344,5461920,22280,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +9087915248,3486688,22300,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +9091405040,2400,22315,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +9091409136,2080,22333,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +9091413200,46048,22373,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +9091462352,3874528,22381,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +9095339248,1952,22384,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +9095343344,2624832,22387,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +9097970928,1856,22398,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +9097975024,33697120,22427,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +9131674896,139840,22468,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +9131817200,11919712,22476,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +9143739600,1952,22479,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +9143743760,10381664,22482,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +9154128944,1952,22493,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +9154133104,1824,22508,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +9154136304,15711680,22529,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +9169850608,5507136,22542,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +9175359728,1920,22550,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +9175363824,1568,22561,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +9175367920,3168,22572,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +9175374096,3392,22586,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +9175380144,1344,22598,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +9175384336,4928,22610,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +9175390736,1120,22623,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +9175394448,1632,22635,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +9175398640,1696,22648,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +9175402768,7616,22659,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +9175412976,26336,22680,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +9175441520,3844544,22703,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +9179287792,2336,22718,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +9179291888,2048,22736,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +9179295984,48768,22776,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +9179347184,3828800,22784,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +9183177968,4032,22787,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +9183184112,2423808,22790,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +9185611024,1888,22801,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +9185615056,24620416,22830,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +9210238192,12466528,22847,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +9222706064,480,22861,,,,,,,,,,0.000,466.667,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +9222707760,2350880,22864,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +9225061616,3725056,22897,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +9228787952,3735968,22899,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +9232525520,5525984,22915,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +9238053104,5802432,22938,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +9243858128,242206816,22941,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +9486066960,2168960,22948,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +9488237840,2016,22951,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +9488241904,2944,22965,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +9488246128,1536,22980,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +9488250096,45504,23003,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +9488297168,2763104,23015,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +9491063024,1920,23024,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +9491067120,7944544,23053,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +9499014416,5450496,23067,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +9504466192,3475360,23087,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +9507942800,2368,23102,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +9507946736,2048,23120,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +9507950832,47552,23160,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +9507999984,3421152,23168,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +9511424272,1952,23171,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +9511428336,2483008,23174,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +9513912656,2368,23185,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +9513916848,34376960,23214,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +9548296464,139616,23255,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +9548438768,12796832,23263,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +9561237712,2240,23266,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +9561241840,10357696,23269,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +9571601648,1920,23280,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +9571605744,832,23295,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +9571607856,15753632,23316,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +9587363056,5458112,23329,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +9592823024,1984,23337,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +9592827216,1408,23348,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +9592831216,1312,23359,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +9592835312,3296,23373,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +9592841456,1344,23385,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +9592845456,4448,23397,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +9592851664,1088,23410,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +9592855696,1632,23422,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +9592859888,1472,23435,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +9592863984,1280,23446,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +9592868080,25184,23467,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +9592894704,3628416,23490,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +9596533904,17728,23505,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +9596554480,11680,23523,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +9596568848,69088,23563,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +9596639472,4118272,23571,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +9600760016,3840,23574,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +9600766224,2419840,23577,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +9603187952,1888,23588,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +9603192048,24930496,23617,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +9628124400,12466368,23634,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +9640592240,544,23648,,,,,,,,,,0.000,411.765,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +9640593680,2350944,23651,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +9642946832,3729024,23684,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +9646678224,3748576,23686,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +9650428112,5354496,23702,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +9655784720,5783808,23725,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +9661571312,241193152,23728,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +9902766384,2168640,23735,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +9904937200,1952,23738,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +9904941296,2944,23752,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +9904945520,1568,23767,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +9904949456,47232,23790,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +9904998640,2753088,23802,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +9907754224,1888,23811,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +9907758320,7844192,23840,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +9915605264,5462816,23854,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +9921070352,3466848,23874,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +9924539632,2304,23889,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +9924543728,2048,23907,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +9924547792,47456,23947,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +9924596944,3416192,23955,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +9928016112,1952,23958,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +9928020208,2424832,23961,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +9930448112,1856,23972,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +9930452208,34503808,24001,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +9964957968,140736,24042,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +9965101264,12747232,24050,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +9977850096,2208,24053,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +9977854192,10374624,24056,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +9988231408,1920,24067,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +9988235504,2528,24082,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +9988239600,15701312,24103,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +10003943664,5469120,24116,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +10009414896,1920,24124,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +10009418992,1408,24135,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +10009423088,1376,24146,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +10009427184,3136,24160,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +10009433328,1344,24172,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +10009437424,6720,24184,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +10009445616,1088,24197,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +10009449712,1600,24209,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +10009453808,1472,24222,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +10009457904,1312,24233,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +10009462000,27584,24254,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +10009492688,3491904,24277,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +10012987664,2528,24292,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +10012991824,2016,24310,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +10012995824,48352,24350,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +10013046000,4232896,24358,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +10017281232,4096,24361,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +10017287376,2498976,24364,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +10019789040,1952,24375,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +10019793104,24911776,24404,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +10044707056,12388768,24421,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +10057100048,416,24435,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +10057102288,2363744,24438,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +10059469072,3784768,24471,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +10063255792,3747296,24473,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +10067004368,5073216,24489,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +10072080656,5698400,24512,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +10077780336,242541280,24515,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +10320322960,2168832,24522,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +10322493680,1920,24525,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +10322497776,2944,24539,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +10322502032,1536,24554,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +10322505968,49696,24577,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +10322557232,2751232,24589,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +10325310704,1952,24598,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +10325314800,7992640,24627,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +10333310288,5463872,24641,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +10338775408,3477760,24661,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +10342254832,2368,24676,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +10342258928,2048,24694,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +10342263024,46048,24734,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +10342312176,3447488,24742,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +10345762032,1952,24745,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +10345766128,2423296,24748,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +10348191984,1920,24759,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +10348196080,34563424,24788,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +10382762288,140576,24829,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +10382905648,12654624,24837,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +10395562192,2688,24840,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +10395566320,10480992,24843,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +10406050032,1920,24854,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +10406054096,1088,24869,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +10406056656,15701088,24890,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +10421760240,5465056,24903,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +10427228432,1920,24911,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +10427232496,1408,24922,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +10427236592,1280,24933,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +10427240688,3232,24947,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +10427246864,1344,24959,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +10427250928,4832,24971,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +10427257040,1120,24984,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +10427261232,1632,24996,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +10427265264,1472,25009,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +10427269360,1280,25020,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +10427273456,25920,25041,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +10427302128,3508896,25064,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +10430812144,2464,25079,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +10430816496,2016,25097,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +10430820592,48512,25137,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +10430871792,3516544,25145,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +10434391280,4384,25148,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +10434397424,2736736,25151,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +10437136624,1888,25162,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +10437140720,25314144,25191,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +10462456176,12035712,25208,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +10474496016,448,25222,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +10474497360,2454048,25225,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +10476953840,3817408,25258,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +10480773360,3902560,25260,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +10484678928,5091520,25276,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +10489772272,5706880,25299,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +10495482096,242566784,25302,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +10738051344,2171840,25309,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +10740226256,1984,25312,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +10740230384,3104,25326,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +10740236496,1536,25341,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +10740240528,47072,25364,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +10740289776,2751712,25376,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +10743043312,1888,25385,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +10743047408,7940800,25414,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +10750989584,5517184,25428,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +10756508912,3821792,25448,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +10760332496,2496,25463,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +10760336720,2048,25481,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +10760340720,46816,25521,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +10760388816,3812448,25529,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +10764204272,1920,25532,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +10764208336,2426976,25535,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +10766637296,1888,25546,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +10766641392,34306016,25575,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +10800949488,140000,25616,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +10801091824,12269024,25624,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +10813362416,5312,25627,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +10813370608,10774496,25630,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +10824147184,4960,25641,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +10824153424,7264,25656,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +10824163568,15771104,25677,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +10839935984,5466784,25690,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +10845405424,1920,25698,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +10845409616,1440,25709,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +10845413648,3168,25720,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +10845419760,3168,25734,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +10845425808,1312,25746,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +10845430000,4544,25758,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +10845436144,1088,25771,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +10845440272,1632,25783,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +10845444336,1472,25796,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +10845448400,1312,25807,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +10845452528,26560,25828,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +10845481200,3503456,25851,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +10848987408,2368,25866,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +10848991472,2016,25884,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +10848995568,48320,25924,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +10849045712,3440576,25932,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +10852487568,4032,25935,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +10852493584,2674176,25938,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +10855170288,1920,25949,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +10855174416,25492928,25978,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +10880668912,11806688,25995,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +10892477296,448,26009,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +10892478928,2395648,26012,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +10894875888,3943904,26045,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +10898821328,3916544,26047,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +10902740176,5236320,26063,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +10907977936,5662944,26086,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +10913642736,242114368,26089,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +11155759408,2167616,26096,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +11157928336,1920,26099,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +11157932400,2912,26113,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +11157938448,1504,26128,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +11157942384,47264,26151,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +11157991632,2756704,26163,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +11160751312,1984,26172,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +11160755472,7890016,26201,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +11168647472,5505440,26215,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +11174154512,3673952,26235,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +11177830640,2592,26250,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +11177834704,2048,26268,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +11177841872,55936,26308,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +11177899216,4036064,26316,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +11181937872,1984,26319,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +11181942000,2423648,26322,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +11184367856,1888,26333,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +11184371952,34814112,26362,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +11219188016,140544,26403,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +11219331312,11914592,26411,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +11231248592,1952,26414,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +11231252720,10789568,26417,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +11242043632,2112,26428,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +11242047696,960,26443,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +11242049968,16312800,26464,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +11258365168,5469024,26477,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +11263835472,2144,26485,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +11263840496,1408,26496,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +11263844592,1280,26507,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +11263848720,3296,26521,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +11263854864,1440,26533,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +11263858928,4320,26545,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +11263865072,1120,26558,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +11263869168,1600,26570,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +11263873264,1504,26583,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +11263877360,1248,26594,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +11263881424,25120,26615,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +11263908080,3509536,26638,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +11267419344,2432,26653,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +11267423472,1984,26671,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +11267427568,48000,26711,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +11267476944,3441024,26719,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +11270920432,3776,26722,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +11270926576,2436416,26725,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +11273364720,1856,26736,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +11273368816,25923936,26765,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +11299295472,11797920,26782,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +11311094640,448,26796,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +11311095984,2385056,26799,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +11313484016,3934176,26832,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +11317421296,3955424,26834,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +11321379536,5190528,26850,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +11326571728,5673920,26873,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +11332247792,242693856,26876,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +11574944016,2174560,26883,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +11577120976,1984,26886,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +11577125104,2752,26900,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +11577129168,1536,26915,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +11577133264,45664,26938,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +11577180368,2752608,26950,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +11579935952,1920,26959,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +11579940048,7985536,26988,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +11587928368,5481344,27002,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +11593411216,3706112,27022,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +11597119728,2560,27037,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +11597123856,2336,27055,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +11597127920,47072,27095,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +11597178096,4002432,27103,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +11601182960,1952,27106,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +11601186800,2423616,27109,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +11603611888,1888,27120,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +11603615984,34343840,27149,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +11637962000,140544,27190,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +11638105328,11934080,27198,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +11650041072,2144,27201,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +11650045168,10771616,27204,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +11660819696,2016,27215,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +11660823792,1056,27230,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +11660827888,16033728,27251,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +11676863728,5465600,27264,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +11682330896,1952,27272,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +11682334992,1440,27283,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +11682337936,1312,27294,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +11682341072,3296,27308,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +11682347248,1312,27320,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +11682351248,4576,27332,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +11682357488,1088,27345,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +11682361584,1632,27357,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +11682365680,1472,27370,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +11682369776,1280,27381,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +11682373872,26848,27402,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +11682402544,3503616,27425,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +11685908720,2496,27440,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +11685912784,1984,27458,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +11685916912,47392,27498,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +11685966064,3441888,27506,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +11689409744,3264,27509,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +11689415920,2435552,27512,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +11691853072,1856,27523,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +11691857136,26544096,27552,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +11718404336,11803168,27569,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +11730208624,416,27583,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +11730210192,2344192,27586,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +11732556080,3879392,27619,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +11736438000,3889536,27621,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +11740330192,5382944,27637,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +11745714416,5676192,27660,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +11751392528,241273408,27663,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +11992667440,2164000,27670,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +11994834128,1952,27673,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +11994838256,2848,27687,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +11994843376,1536,27702,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +11994847344,47232,27725,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +11994896592,2758016,27737,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +11997657328,2080,27746,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +11997661392,7871520,27775,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +12005536048,5454944,27789,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +12010993904,3469600,27809,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +12014465264,2304,27824,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12014469360,2080,27842,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12014473456,46176,27882,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12014522608,3422656,27890,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +12017946864,1952,27893,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +12017950960,2425152,27896,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +12020378864,1888,27907,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +12020382992,34162048,27936,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +12054547696,139200,27977,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12054688272,12332832,27985,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +12067024112,1952,27988,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +12067028208,10806528,27991,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +12077837552,1920,28002,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +12077841616,3072,28017,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12077847792,15536672,28038,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +12093387024,5629920,28051,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +12099018992,1920,28059,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +12099023152,1408,28070,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +12099027184,1472,28081,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +12099031280,3360,28095,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +12099036400,1312,28107,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +12099040496,4448,28119,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +12099046640,1184,28132,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +12099050704,1824,28144,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +12099054832,1632,28157,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +12099058928,1280,28168,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12099063024,25696,28189,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +12099091696,3543520,28212,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +12102636784,2464,28227,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12102640880,2304,28245,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12102644944,63616,28285,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12102710480,3828128,28293,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +12106541296,3296,28296,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +12106547440,2443872,28299,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +12108993776,1888,28310,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +12108997872,24936000,28339,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +12133935344,11814080,28356,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +12145751952,416,28370,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +12145753296,2352896,28373,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +12148107472,3725216,28406,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +12151834832,3740704,28408,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +12155578576,5090944,28424,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +12160671984,5663328,28447,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +12166337776,241822912,28450,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +12408162608,2157568,28457,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +12410323184,1952,28460,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +12410327280,2976,28474,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +12410331568,1504,28489,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +12410335216,50112,28512,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12410386704,2750304,28524,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +12413139184,1920,28533,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +12413143280,7919840,28562,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +12421064976,5455904,28576,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +12426523888,3468640,28596,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +12429995248,2368,28611,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12429999344,1984,28629,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12430003408,48128,28669,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12430054640,3424576,28677,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +12433481968,1952,28680,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +12433486096,2426240,28683,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +12435913968,1856,28694,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +12435918064,33735648,28723,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +12469656848,140768,28764,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12469799152,12803680,28772,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +12482604336,2176,28775,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +12482608368,10379136,28778,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +12492989680,1952,28789,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +12492993776,864,28804,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12492995920,15888288,28825,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +12508887280,5504480,28838,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +12514393328,2112,28846,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +12514397424,1408,28857,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +12514401520,1280,28868,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +12514405616,3744,28882,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +12514411728,1344,28894,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +12514414544,4896,28906,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +12514421904,1088,28919,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +12514426096,1632,28931,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +12514430192,1760,28944,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +12514434288,1248,28955,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12514438352,34208,28976,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +12514475248,3824608,28999,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +12518302960,2528,29014,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12518307056,2016,29032,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12518311152,47136,29072,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12518360272,3841696,29080,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +12522203344,3968,29083,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +12522209552,2437024,29086,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +12524648688,1888,29097,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +12524652816,24963488,29126,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +12549617904,12417312,29143,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +12562036624,416,29157,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +12562038256,2346336,29160,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +12564387056,3717536,29193,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +12568106224,3741536,29195,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +12571849040,5525152,29211,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +12577376496,5689024,29234,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +12583066864,242627456,29237,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +12825695632,2167232,29244,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +12827865328,1984,29247,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +12827869456,2976,29261,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +12827873744,1504,29276,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +12827877616,47264,29299,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12827926736,2752640,29311,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +12830681328,1952,29320,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +12830685424,7922240,29349,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +12838610192,5462688,29363,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +12844074192,3470944,29383,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +12847546608,2432,29398,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12847550704,2112,29416,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12847554800,47168,29456,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12847604976,3418752,29464,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +12851025136,1952,29467,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +12851029232,2467008,29470,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +12853498096,2080,29481,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +12853502192,34443328,29510,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +12887948560,140224,29551,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12888091888,11968896,29559,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +12900063440,2208,29562,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +12900067536,10367680,29565,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +12910436592,1952,29576,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +12910440688,832,29591,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12910442800,15742528,29612,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +12926187760,5472992,29625,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +12931662032,1984,29633,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +12931666160,1440,29644,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +12931670256,3712,29655,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +12931676400,3456,29669,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +12931682448,1344,29681,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +12931686640,4352,29693,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +12931692784,1120,29706,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +12931696880,1664,29718,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +12931700976,1504,29731,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +12931705136,1280,29742,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12931709168,25920,29763,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +12931737936,3556960,29786,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +12935297264,2464,29801,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12935301360,2208,29819,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12935305456,48480,29859,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12935355760,4213088,29867,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +12939571472,3584,29870,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +12939577584,2431296,29873,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +12942010640,1888,29884,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +12942014704,24939456,29913,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +12966956272,12481600,29930,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +12979440560,448,29944,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +12979441872,2345504,29947,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +12981789936,3715584,29980,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +12985507024,3743616,29982,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +12989252848,5140192,29998,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +12994395376,5886496,30021,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +13000284464,243240192,30024,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +13243527472,2167936,30031,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +13245698288,1952,30034,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +13245702352,2784,30048,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +13245706480,1536,30063,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +13245710576,47040,30086,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +13245760720,2864192,30098,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +13248626928,1888,30107,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +13248631024,7919136,30136,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +13256551696,5460800,30150,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +13262013808,3467264,30170,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +13265482992,2336,30185,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +13265487056,2080,30203,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +13265491184,46144,30243,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +13265540336,3419168,30251,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +13268962512,1984,30254,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +13268966640,2430464,30257,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +13271399664,1888,30268,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +13271403760,34480480,30297,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +13305886992,140800,30338,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +13306030064,11942176,30346,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +13317974224,2464,30349,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +13317978352,10435040,30352,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +13328414960,1920,30363,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +13328419184,896,30378,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +13328423184,15770560,30399,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +13344196880,5478400,30412,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +13349678320,1920,30420,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +13349682416,1440,30431,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +13349686480,1344,30442,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +13349690576,3424,30456,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +13349696752,1344,30468,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +13349700848,4288,30480,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +13349706992,1088,30493,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +13349711088,1632,30505,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +13349715184,1472,30518,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +13349719312,1440,30529,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +13349723344,25152,30550,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +13349750000,3505472,30573,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +13353257200,2432,30588,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +13353261296,1984,30606,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +13353265360,47424,30646,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +13353314512,4290496,30654,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +13357609776,5088,30657,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +13357617392,2470944,30660,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +13360091376,1920,30671,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +13360095472,24951584,30700,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +13385048336,12361888,30717,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +13397412144,416,30731,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +13397413936,2352704,30734,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +13399769328,3794400,30767,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +13403566320,3743264,30769,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +13407311056,5097024,30785,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +13412409680,5794272,30808,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +13418206448,242996160,30811,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +13661205776,2172704,30818,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +13663380688,1984,30821,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +13663384816,2752,30835,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +13663388912,1536,30850,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +13663393008,45984,30873,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +13663440240,2770784,30885,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +13666213072,2016,30894,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +13666217264,7897696,30923,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +13674116368,5469888,30937,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +13679587536,3478528,30957,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +13683068144,2336,30972,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +13683072336,2176,30990,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +13683076528,47104,31030,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +13683125584,3417632,31038,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +13686544592,1952,31041,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +13686548720,2435296,31044,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +13688985872,1920,31055,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +13688989904,35071936,31084,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +13724064048,140832,31125,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +13724206320,11934080,31133,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +13736143184,2208,31136,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +13736147184,10473696,31139,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +13746623728,1920,31150,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +13746627792,864,31165,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +13746629936,15780192,31186,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +13762411760,5487200,31199,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +13767901520,2016,31207,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +13767905616,1440,31218,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +13767909616,1312,31229,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +13767913808,3488,31243,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +13767919856,1312,31255,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +13767923952,4704,31267,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +13767930192,1120,31280,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +13767934224,1632,31292,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +13767938384,1536,31305,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +13767942480,1312,31316,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +13767946480,26112,31337,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +13767975152,3522240,31360,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +13771499728,2336,31375,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +13771503856,1984,31393,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +13771507952,47360,31433,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +13771557488,3504160,31441,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +13775063312,3200,31444,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +13775069456,2750464,31447,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +13777822128,1888,31458,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +13777826032,25831808,31487,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +13803659536,12180320,31504,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +13815841840,448,31518,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +13815843056,2370304,31521,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +13818215664,3851904,31554,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +13822070000,3824896,31556,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +13825897680,5097152,31572,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +13830996208,5708416,31595,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +13836707152,243821344,31598,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +14080530704,2179040,31605,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +14082711760,1920,31608,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +14082715984,2752,31622,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +14082720112,1632,31637,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +14082724080,49440,31660,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +14082775280,2768608,31672,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +14085545200,1888,31681,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +14085549296,8041856,31710,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +14093592880,5523488,31724,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +14099119312,3508768,31744,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +14102629712,2336,31759,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +14102633712,2016,31777,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +14102637936,45984,31817,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +14102685936,3415424,31825,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +14106104016,1952,31828,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +14106108144,2435424,31831,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +14108546288,1888,31842,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +14108550384,34749824,31871,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +14143302928,139584,31912,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +14143444176,12637024,31920,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +14156083440,2208,31923,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +14156087536,10553792,31926,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +14166643952,1920,31937,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +14166648048,2944,31952,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +14166652272,15802976,31973,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +14182456528,5484832,31986,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +14187943152,1952,31994,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +14187947248,1440,32005,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +14187951440,1632,32016,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +14187955440,3232,32030,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +14187961744,1376,32042,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +14187965776,4896,32054,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +14187971984,1344,32067,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +14187976016,1632,32079,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +14187979984,1504,32092,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +14187984112,1280,32103,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +14187988432,28864,32124,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +14188018928,3517408,32147,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +14191538512,2496,32162,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +14191542512,1952,32180,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +14191546704,49504,32220,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +14191598928,3880416,32228,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +14195480816,3968,32231,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +14195486960,2623360,32234,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +14198112592,1888,32245,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +14198116592,25732096,32274,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +14223850736,12178304,32291,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +14236030896,384,32305,,,,,,,,,,0.000,583.333,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +14236032496,2502656,32308,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +14238537744,3782816,32341,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +14242322640,3773120,32343,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +14246097104,5226624,32359,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +14251325680,5718816,32382,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +14257045808,245704832,32385,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +14502752656,2174368,32392,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +14504929488,1984,32395,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +14504933616,2944,32409,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +14504937872,1536,32424,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +14504941808,47296,32447,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +14504990928,2773440,32459,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +14507766096,1952,32468,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +14507770224,8355648,32497,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +14516129136,5595744,32511,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +14521727248,3582016,32531,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +14525311184,2336,32546,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +14525315312,2016,32564,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +14525319504,45600,32604,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +14525366544,3429056,32612,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +14528798032,1984,32615,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +14528802032,2428384,32618,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +14531231984,1984,32629,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +14531236112,34769312,32658,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +14566007120,140896,32699,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +14566150480,12833760,32707,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +14578987216,1952,32710,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +14578991344,10459904,32713,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +14589452592,1952,32724,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +14589456624,864,32739,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +14589459120,15787264,32760,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +14605248752,5485600,32773,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +14610736496,1952,32781,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +14610740464,1472,32792,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +14610744656,1312,32803,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +14610748752,3552,32817,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +14610754800,1344,32829,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +14610758992,4800,32841,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +14610765072,1120,32854,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +14610769232,1632,32866,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +14610773328,1472,32879,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +14610777328,1376,32890,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +14610781520,27456,32911,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +14610810256,3513824,32934,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +14614326608,2656,32949,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +14614330608,1952,32967,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +14614334800,54208,33007,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +14614392048,4344608,33015,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +14618737968,4000,33018,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +14618744144,2452704,33021,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +14621198608,1856,33032,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +14621202672,25020704,33061,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +14646225104,12401824,33078,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +14658628496,416,33092,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +14658629808,2358784,33095,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +14660991216,3720512,33128,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +14664713552,3752288,33130,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +14668467408,5166208,33146,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +14673636688,5912992,33169,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +14679552240,243265760,33172,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +14922819888,2181408,33179,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +14925003024,1984,33182,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +14925007088,2944,33196,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +14925011344,1504,33211,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +14925015280,45504,33234,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +14925063376,2772544,33246,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +14927838448,1888,33255,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +14927842512,7908448,33284,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +14935753104,5470688,33298,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +14941226224,3486752,33318,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +14944716112,2336,33333,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +14944720112,1952,33351,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +14944724304,46944,33391,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +14944774576,3429888,33399,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +14948205840,1952,33402,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +14948209872,2433824,33405,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +14950645008,1856,33416,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +14950649168,34601024,33445,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +14985252144,140448,33486,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +14985394416,12787328,33494,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +14998183120,1952,33497,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +14998187248,10507648,33500,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +15008697584,1920,33511,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +15008701808,832,33526,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +15008704208,15794528,33547,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +15024500976,5490432,33560,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +15029992720,2080,33568,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +15029996816,1440,33579,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +15030000912,3392,33590,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +15030007024,3488,33604,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +15030013168,1312,33616,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +15030017264,4608,33628,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +15030023376,1120,33641,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +15030027504,1664,33653,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +15030031600,1472,33666,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +15030035696,1280,33677,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +15030039792,26944,33698,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +15030068464,3510464,33721,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +15033580752,2432,33736,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +15033584848,2016,33754,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +15033588944,47936,33794,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +15033639152,4327456,33802,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +15037968656,3584,33805,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +15037974800,2438464,33808,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +15040415984,1856,33819,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +15040420080,24980000,33848,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +15065401712,12355584,33865,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +15077759920,416,33879,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +15077761520,2391040,33882,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +15080155472,3754496,33915,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +15083912400,3751616,33917,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +15087665392,5103328,33933,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +15092770000,5872896,33956,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +15098644816,243853824,33959,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +15342500112,2173248,33966,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +15344675024,1920,33969,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +15344679152,2944,33983,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +15344683376,1504,33998,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +15344687344,45984,34021,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +15344734640,2776768,34033,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +15347512720,1888,34042,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +15347516656,7939712,34071,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +15355458864,5468864,34085,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +15360929008,3482720,34105,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +15364413712,2432,34120,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +15364417776,2144,34138,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +15364421872,46048,34178,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +15364470000,3435008,34186,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +15367906544,1920,34189,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +15367910640,2439200,34192,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +15370351856,1920,34203,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +15370355952,34622048,34232,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +15404980528,142464,34273,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +15405124944,12814048,34281,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +15417941232,1984,34284,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +15417945328,10463552,34287,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +15428410704,1952,34298,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +15428414800,864,34313,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +15428418832,15803776,34334,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +15444225264,5480064,34347,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +15449707856,1920,34355,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +15449711952,1472,34366,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +15449715952,1312,34377,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +15449720144,3552,34391,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +15449726192,1344,34403,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +15449730288,4480,34415,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +15449736592,1088,34428,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +15449740496,1632,34440,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +15449744848,1504,34453,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +15449748816,1280,34464,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +15449752816,25952,34485,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +15449781488,3512480,34508,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +15453295952,2400,34523,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +15453299920,2176,34541,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +15453304144,47104,34581,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +15453354256,4206304,34589,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +15457562896,3264,34592,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +15457569040,2482784,34595,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +15460054352,1888,34606,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +15460058448,24982208,34635,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +15485041936,12379232,34652,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +15497423760,416,34666,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +15497425328,2372352,34669,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +15499799792,3764256,34702,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +15503566032,3745824,34704,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +15507314896,5102048,34720,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +15512418640,5740512,34743,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +15518161904,243390656,34746,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +15761553872,2179552,34753,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +15763735760,1984,34756,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +15763739888,2848,34770,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +15763744048,1504,34785,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +15763748048,46464,34808,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +15763796304,2774592,34820,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +15766573296,1920,34829,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +15766577392,7937408,34858,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +15774517552,5466464,34872,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +15779985648,3486240,34892,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +15783474512,2400,34907,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +15783478608,2016,34925,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +15783482608,45856,34965,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +15783530864,3417632,34973,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +15786949840,1952,34976,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +15786953968,2433792,34979,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +15789390064,1856,34990,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +15789394160,34660288,35019,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +15824057616,140576,35060,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +15824200944,12779424,35068,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +15836982480,2496,35071,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +15836986608,10441280,35074,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +15847429456,2016,35085,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +15847433552,2912,35100,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +15847439600,15823232,35121,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +15863264496,5483136,35134,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +15868750192,1920,35142,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +15868754160,1440,35153,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +15868758352,3424,35164,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +15868764368,3296,35178,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +15868770640,1344,35190,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +15868774768,4416,35202,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +15868780848,1088,35215,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +15868785008,1632,35227,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +15868788976,1504,35240,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +15868793072,1280,35251,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +15868797296,26336,35272,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +15868825840,3516480,35295,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +15872345328,2336,35310,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +15872349424,2016,35328,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +15872353520,47872,35368,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +15872402736,3900736,35376,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +15876305232,1952,35379,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +15876309232,2704032,35382,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +15879014608,2208,35393,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +15879018736,25014784,35422,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +15904036048,12223424,35439,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +15916260400,416,35453,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +15916262000,2486112,35456,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +15918750960,3770784,35489,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +15922524400,3784096,35491,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +15926311152,5109984,35507,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +15931423984,5723136,35530,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +15937149168,243361312,35533,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +16180513040,2178944,35540,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +16182694224,1952,35543,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +16182698320,3200,35557,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +16182704368,1536,35572,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +16182708560,49120,35595,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +16182760752,2770976,35607,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +16185534736,1920,35616,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +16185538896,8428352,35645,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +16193968624,5499328,35659,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +16199469424,3676128,35679,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +16203147504,2496,35694,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +16203151600,2080,35712,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +16203155664,47648,35752,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +16203204880,4130240,35760,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +16207336816,1952,35763,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +16207340784,2429920,35766,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +16209773776,1888,35777,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +16209778000,34611488,35806,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +16244391184,139936,35847,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +16244532464,12928448,35855,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +16257463504,2208,35858,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +16257467632,10450016,35861,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +16267919600,1920,35872,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +16267923792,832,35887,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +16267926288,15834848,35908,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +16283762992,5478816,35921,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +16289244400,1952,35929,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +16289248496,1408,35940,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +16289251600,1472,35951,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +16289254608,3456,35965,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +16289260784,1376,35977,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +16289264880,4800,35989,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +16289270992,1088,36002,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +16289275120,1600,36014,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +16289279216,1504,36027,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +16289283312,1280,36038,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +16289287408,26624,36059,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +16289316080,3510240,36082,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +16292829520,2432,36097,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +16292833488,2048,36115,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +16292837616,48736,36155,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +16292887760,4145728,36163,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +16297036048,3808,36166,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +16297042192,2516704,36169,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +16299560208,1888,36180,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +16299564272,25347424,36209,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +16324914416,12396480,36226,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +16337312656,768,36240,,,,,,,,,,0.000,291.667,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +16337314512,2354336,36243,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +16339671280,3785536,36276,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +16343459056,3750080,36278,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +16347211952,5090848,36294,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +16352304368,5793280,36317,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +16358099248,242750816,36320,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +16600852144,2176192,36327,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +16603030864,1984,36330,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +16603034864,3008,36344,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +16603041008,1536,36359,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +16603045072,48992,36382,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +16603096272,2770880,36394,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +16605869296,2016,36403,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +16605873392,8377760,36432,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +16614253808,5504512,36446,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +16619759856,3685984,36466,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +16623447376,2432,36481,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +16623451344,2176,36499,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +16623455568,47456,36539,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +16623504624,3658144,36547,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +16627165424,1952,36550,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +16627169520,2434048,36553,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +16629605616,1920,36564,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +16629609712,34581888,36593,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +16664193296,141376,36634,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +16664336592,12743040,36642,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +16677082320,2208,36645,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +16677086416,10444640,36648,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +16687532336,1952,36659,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +16687536368,928,36674,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +16687538576,15807712,36695,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +16703349008,5580608,36708,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +16708931824,1920,36716,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +16708936080,1440,36727,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +16708940080,1312,36738,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +16708946000,22080,36752,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +16709056560,1344,36764,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +16709059824,5952,36776,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +16709068016,1120,36789,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +16709072240,6144,36801,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +16709210704,1952,36814,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +16709215472,1248,36825,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +16709219536,38432,36846,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +16709259504,3551520,36869,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +16712812752,5216,36884,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +16712820944,2432,36902,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +16712825072,49920,36942,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +16712877296,4215904,36950,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +16717096144,4128,36953,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +16717102320,2539872,36956,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +16719643984,1856,36967,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +16719647952,24050080,36996,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +16743699728,12171808,37013,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +16755873680,416,37027,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +16755875344,2411200,37030,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +16758304432,3849120,37063,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +16762155248,3791648,37065,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +16765949136,5113664,37081,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +16771065168,5721440,37104,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +16776789232,243282592,37107,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +17020074352,2181056,37114,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +17022257456,1952,37117,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +17022261488,2752,37131,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +17022265680,1536,37146,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +17022268848,47200,37169,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +17022317776,2768704,37181,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +17025088048,1888,37190,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +17025091824,7918304,37219,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +17033012496,5477536,37233,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +17038493008,3490304,37253,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +17041984752,2592,37268,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +17041988848,1952,37286,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +17041992944,47552,37326,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +17042043088,3416192,37334,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +17045461200,2048,37337,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +17045465328,2434656,37340,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +17047902576,1920,37351,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +17047906512,34610912,37380,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +17082518832,140352,37421,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +17082661136,11964480,37429,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +17094628656,2208,37432,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +17094632656,10482784,37435,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +17105117776,1920,37446,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +17105121552,864,37461,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +17105123696,15782656,37482,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +17120908528,5489472,37495,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +17126400240,1984,37503,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +17126404336,1440,37514,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +17126408848,3520,37525,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +17126414544,3456,37539,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +17126420816,1344,37551,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +17126424816,4448,37563,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +17126431056,1120,37576,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +17126435184,1632,37588,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +17126439152,1472,37601,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +17126443376,1248,37612,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +17126447440,26944,37633,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +17126476080,3512928,37656,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +17129991504,2400,37671,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +17129995504,2080,37689,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +17129999824,47872,37729,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +17130050128,3447936,37737,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +17133499728,3616,37740,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +17133505872,2833664,37743,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +17136341232,1888,37754,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +17136345424,25380896,37783,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +17161729232,11852064,37800,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +17173582736,448,37814,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +17173584624,2479616,37817,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +17176067312,3863872,37850,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +17179932912,3950816,37852,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +17183885520,5209696,37868,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +17189096784,5707104,37891,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +17194805488,243530720,37894,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +17438338320,2179584,37901,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +17440520400,2048,37904,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +17440524528,2752,37918,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +17440528592,1536,37933,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +17440532688,45664,37956,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +17440580848,2772320,37968,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +17443354864,1888,37977,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +17443358960,7933664,38006,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +17451293936,5481440,38020,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +17456777424,3490848,38040,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +17460270320,2304,38055,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +17460274416,2176,38073,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +17460278544,46048,38113,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +17460327760,3426208,38121,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +17463757008,1952,38124,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +17463761136,2438752,38127,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +17466202352,1888,38138,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +17466206448,34688000,38167,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +17500896528,140192,38208,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +17501039856,12220320,38216,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +17513261552,2016,38219,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +17513265424,10986432,38222,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +17524253936,1952,38233,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +17524258032,896,38248,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +17524260208,15775616,38269,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +17540037872,5476288,38282,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +17545516272,1952,38290,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +17545520368,1440,38301,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +17545524464,1280,38312,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +17545528560,3040,38326,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +17545534704,1440,38338,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +17545537392,4672,38350,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +17545544016,1088,38363,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +17545548016,1600,38375,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +17545552112,1504,38388,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +17545556336,1248,38399,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +17545560304,25984,38420,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +17545588976,3520064,38443,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +17549110512,2688,38458,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +17549114608,2016,38476,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +17549118672,49440,38516,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +17549169904,3445856,38524,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +17552617808,3680,38527,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +17552623824,2677920,38530,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +17555303760,1888,38541,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +17555307856,25633504,38570,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +17580943632,11815744,38587,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +17592761232,416,38601,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +17592762832,2383392,38604,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +17595147504,3935872,38637,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +17599085808,3919136,38639,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +17603006704,5242336,38655,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +17608250704,5685216,38678,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +17613938928,243162464,38681,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +17857103120,2174272,38688,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +17859280144,1920,38691,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +17859284208,2624,38705,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +17859288272,1536,38720,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +17859292400,45184,38743,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +17859339472,2773504,38755,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +17862114256,1920,38764,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +17862118672,7919360,38793,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +17870040464,5467552,38807,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +17875510480,3489920,38827,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +17879003440,2400,38842,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +17879007536,1952,38860,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +17879011568,47104,38900,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +17879061744,3419008,38908,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +17882482064,1984,38911,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +17882486096,2434304,38914,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +17884923120,1920,38925,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +17884927216,34663328,38954,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +17919591888,140288,38995,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +17919735024,11913600,39003,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +17931650256,2176,39006,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +17931654384,10436992,39009,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +17942094064,1920,39020,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +17942098192,2720,39035,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +17942102288,15837408,39056,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +17957941488,5479968,39069,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +17963422960,1920,39077,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +17963427152,1408,39088,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +17963431248,1312,39099,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +17963435216,3104,39113,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +17963441488,1376,39125,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +17963445488,4480,39137,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +17963451472,1088,39150,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +17963454352,1632,39162,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +17963458800,1472,39175,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +17963462896,1280,39186,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +17963466992,25376,39207,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +17963493648,3513056,39230,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +17967008016,2592,39245,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +17967012080,1984,39263,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +17967016176,48640,39303,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +17967067376,3444384,39311,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +17970514288,3840,39314,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +17970520336,2452608,39317,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +17972974864,1856,39328,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +17972978928,25941344,39357,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +17998922992,11825184,39374,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +18010749936,448,39388,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +18010751280,2374368,39391,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +18013127920,3870592,39424,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +18017000720,3897536,39426,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +18020900080,5456384,39442,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +18026358000,5682784,39465,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +18032043216,243159328,39468,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +18275205520,2185664,39475,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +18277393616,1984,39478,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +18277397744,2976,39492,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +18277402000,1536,39507,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +18277405936,50944,39530,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +18277458160,2771296,39542,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +18280232144,1856,39551,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +18280236272,7895424,39580,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +18288133392,5527488,39594,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +18293662928,3679488,39614,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +18297345264,2752,39629,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +18297349328,2240,39647,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +18297353424,48032,39687,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +18297403984,4093952,39695,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +18301500688,1920,39698,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +18301504784,2438080,39701,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +18303944944,1888,39712,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +18303949040,34779168,39741,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +18338731280,141376,39782,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +18338875696,11911488,39790,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +18350788816,2272,39793,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +18350792944,10935616,39796,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +18361731312,7584,39807,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +18361754960,8672,39822,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +18361766224,16098720,39843,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +18377867504,5481952,39856,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +18383352112,1952,39864,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +18383356144,1440,39875,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +18383360304,1312,39886,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +18383364336,3648,39900,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +18383370480,1312,39912,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +18383374544,4896,39924,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +18383380720,1120,39937,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +18383383440,1632,39949,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +18383387888,1472,39962,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +18383392080,1472,39973,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +18383396176,26816,39994,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +18383424752,3515712,40017,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +18386943312,2400,40032,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +18386947376,2048,40050,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +18386951408,49056,40090,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +18387002608,3425184,40098,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +18390430032,3808,40101,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +18390436048,2433728,40104,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +18392871280,1856,40115,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +18392875248,25886688,40144,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +18418765040,11822752,40161,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +18430588816,448,40175,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +18430590416,2358144,40178,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +18432951536,3870208,40211,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +18436824304,3901312,40213,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +18440727888,5449536,40229,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +18446178800,5678048,40252,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +18451859696,244356320,40255,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +18696218928,2190848,40262,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +18698411216,1952,40265,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +18698415440,2944,40279,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +18698421488,1536,40294,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +18698425584,49888,40317,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +18698477904,2770112,40329,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +18701249872,1888,40338,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +18701253616,8399808,40367,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +18709654800,5519104,40381,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +18715176304,3799840,40401,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +18718978288,4928,40416,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +18718984592,2112,40434,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +18718988528,46272,40474,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +18719036624,3862208,40482,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +18722901232,1952,40485,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +18722905328,2437024,40488,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +18725343664,1888,40499,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +18725347600,34519392,40528,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +18759869744,140416,40569,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +18760013040,11940640,40577,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +18771954928,2400,40580,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +18771959120,11079968,40583,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +18783040752,2496,40594,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +18783044848,2144,40609,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +18783048944,15949184,40630,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +18798999792,5503712,40643,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +18804504816,1920,40651,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +18804509040,1440,40662,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +18804513104,1312,40673,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +18804517104,3456,40687,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +18804523344,1440,40699,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +18804527344,4800,40711,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +18804533712,1088,40724,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +18804537680,1632,40736,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +18804541648,1504,40749,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +18804544560,1280,40760,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +18804548176,25920,40781,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +18804576560,3549088,40804,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +18808128752,2400,40819,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +18808132912,2176,40837,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +18808136944,48352,40877,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +18808187120,3443040,40885,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +18811631824,3840,40888,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +18811638000,2447008,40891,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +18814087408,1920,40902,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +18814091600,25021376,40931,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +18839116144,11827520,40948,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +18850945936,448,40962,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +18850947568,2414464,40965,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +18853365072,3933056,40998,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +18857301200,3838272,41000,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +18861141200,5452544,41016,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +18866595056,5752064,41039,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +18872348944,242752640,41042,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +19115104560,2175072,41049,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +19117281520,1952,41052,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +19117285712,2976,41066,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +19117291728,1632,41081,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +19117295824,49280,41104,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +19117347024,2774432,41116,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +19120123120,1888,41125,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +19120127216,7918048,41154,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +19128047888,5545440,41168,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +19133595888,3704416,41188,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +19137301744,6656,41203,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +19137309936,5088,41221,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +19137318096,75392,41261,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +19137394928,3973760,41269,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +19141371120,1952,41272,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +19141375216,2437984,41275,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +19143814576,1888,41286,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +19143818480,34589248,41315,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +19178409232,140128,41356,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +19178551536,11918368,41364,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +19190471888,2176,41367,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +19190475984,10877600,41370,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +19201355088,2048,41381,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +19201359088,864,41396,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +19201361232,16805760,41417,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +19218169072,5480000,41430,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +19223650544,1952,41438,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +19223654640,1408,41449,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +19223658736,3744,41460,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +19223664848,3456,41474,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +19223671056,1344,41486,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +19223675120,4448,41498,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +19223681264,1088,41511,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +19223685328,1664,41523,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +19223689456,1472,41536,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +19223693552,1280,41547,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +19223696560,27104,41568,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +19223726320,3516032,41591,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +19227244784,2400,41606,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +19227248880,2016,41624,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +19227253008,48160,41664,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +19227303248,3527936,41672,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +19230833904,3616,41675,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +19230840048,2491008,41678,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +19233332464,1856,41689,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +19233336560,26017728,41718,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +19259356400,11800000,41735,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +19271158704,416,41749,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +19271160272,2401824,41752,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +19273563376,3883328,41785,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +19277448432,3884192,41787,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +19281334928,5423936,41803,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +19286760688,5687360,41826,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +19292451056,241628288,41829,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +19534082320,2169824,41836,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +19536254192,1952,41839,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +19536258288,2944,41853,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +19536262512,1536,41868,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +19536266480,46432,41891,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +19536315600,2777056,41903,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +19539095792,1920,41912,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +19539099888,7852032,41941,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +19546954992,5480384,41955,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +19552438512,3482688,41975,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +19555924304,2336,41990,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +19555928496,2048,42008,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +19555932496,47680,42048,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +19555981552,3432384,42056,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +19559416048,1952,42059,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +19559420208,2433024,42062,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +19561855312,1888,42073,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +19561859440,34484864,42102,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +19596345616,140064,42143,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +19596487920,12103776,42151,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +19608593712,2048,42154,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +19608597744,10852384,42157,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +19619452144,2048,42168,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +19619456240,832,42183,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +19619458480,15941600,42204,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +19635402992,5533120,42217,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +19640937712,2560,42225,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +19640941808,1440,42236,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +19640946160,1312,42247,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +19640950032,9408,42261,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +19640962320,1408,42273,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +19640966384,4544,42285,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +19640972624,1088,42298,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +19640976624,11488,42310,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +19640990928,10656,42323,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +19641003248,1760,42334,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +19641007472,34688,42355,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +19641045328,3517632,42378,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +19644564720,2368,42393,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +19644568912,2080,42411,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +19644572912,50432,42451,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +19644625136,3448832,42459,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +19648076016,3360,42462,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +19648082160,2457216,42465,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +19650541936,1888,42476,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +19650545904,25767264,42505,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +19676315888,11886144,42522,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +19688203184,416,42536,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +19688204592,2350784,42539,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +19690556688,3769696,42572,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +19694329168,4067488,42574,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +19698399568,5439008,42590,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +19703841104,5761760,42613,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +19709604144,241841472,42616,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +19951447312,2185280,42623,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +19953635664,1952,42626,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +19953639664,2976,42640,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +19953644016,1536,42655,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +19953647952,47104,42678,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +19953698000,2773376,42690,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +19956474096,1856,42699,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +19956478192,7878016,42728,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +19964358928,5479584,42742,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +19969840464,3491872,42762,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +19973335248,2624,42777,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +19973339344,2048,42795,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +19973343440,46688,42835,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +19973391600,4083584,42843,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +19977477392,2336,42846,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +19977481552,2549760,42849,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +19980033264,1888,42860,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +19980037360,34247328,42889,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +20014287120,146144,42930,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +20014434512,12350112,42938,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +20026787056,2208,42941,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +20026791152,10440128,42944,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +20037232912,1920,42955,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +20037236976,3008,42970,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +20037243120,14879520,42991,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +20052124912,5635680,43004,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +20057762000,1984,43012,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +20057766128,1440,43023,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +20057770256,1440,43034,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +20057774320,3072,43048,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +20057780464,1792,43060,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +20057784688,4544,43072,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +20057790704,7584,43085,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +20057800944,1952,43097,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +20057805008,2016,43110,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +20057810320,1248,43121,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20057813232,25632,43142,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +20057841904,3676192,43165,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +20061521136,2400,43180,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20061525328,2240,43198,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20061529488,49472,43238,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +20061581552,3885408,43246,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +20065468624,3648,43249,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +20065474896,2445824,43252,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +20067923184,1888,43263,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +20067927248,25171104,43292,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +20093101264,12445568,43309,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +20105547888,416,43323,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +20105549488,2355424,43326,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +20107906704,3719296,43359,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +20111627600,3800608,43361,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +20115430608,5700544,43377,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +20121133392,5727296,43400,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +20126863664,242556320,43403,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +20369421616,2181824,43410,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +20371604816,1952,43413,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +20371608816,2976,43427,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +20371613072,1504,43442,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +20371616976,53440,43465,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +20371673328,2776864,43477,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +20374451472,1888,43486,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +20374455536,7782656,43515,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +20382240016,5478016,43529,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +20387720432,3498272,43549,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +20391220464,2336,43564,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20391224656,2208,43582,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20391228496,46560,43622,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +20391277808,3830208,43630,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +20395110608,2208,43633,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +20395114736,2680512,43636,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +20397796560,1888,43647,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +20397801008,33924352,43676,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +20431726896,141056,43717,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +20431869200,11924128,43725,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +20443795792,1952,43728,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +20443799824,10514592,43731,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +20454316304,1920,43742,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +20454320368,864,43757,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +20454322608,15693888,43778,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +20470018288,5522272,43791,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +20475542768,2080,43799,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +20475546864,1408,43810,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +20475550960,1312,43821,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +20475555056,10752,43835,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +20475567344,1376,43847,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +20475571440,4672,43859,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +20475577648,1312,43872,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +20475581712,1632,43884,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +20475585776,1600,43897,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +20475590064,1312,43908,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20475593968,31712,43929,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +20475628752,3834528,43952,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +20479465808,2688,43967,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20479470800,2144,43985,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20479474896,49376,44025,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +20479526128,3812864,44033,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +20483340528,3968,44036,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +20483346672,2433376,44039,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +20485782768,1888,44050,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +20485786832,25033120,44079,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +20510821616,12471040,44096,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +20523294608,416,44110,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] +20523295920,2368672,44113,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +20525667568,3739904,44146,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +20529409264,3745888,44148,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +20533157104,5564416,44164,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +20538723632,5812416,44187,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +20544537840,240774464,44190,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +20785314288,2176096,44197,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +20787493328,1920,44200,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" +20787497296,3328,44214,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +20787503312,1664,44229,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +20787507536,47328,44252,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +20787557584,2772384,44264,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +20790331760,1856,44273,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +20790335760,7745440,44302,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +20798084368,5472320,44316,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +20803558640,3491552,44336,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +20807052528,2336,44351,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20807056816,2112,44369,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20807060720,47616,44409,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +20807109840,3424352,44417,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +20810536176,1920,44420,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +20810540272,2456224,44423,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +20812997872,2112,44434,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +20813001968,34352224,44463,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +20847357200,142208,44504,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +20847501648,12776704,44512,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +20860280016,2208,44515,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +20860284240,10460288,44518,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +20870746352,1920,44529,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +20870750448,832,44544,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +20870752592,15672768,44565,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +20886427888,5476768,44578,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel +20891906288,1952,44586,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +20891910384,1440,44597,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +20891914480,3648,44608,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +20891920592,3456,44622,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +20891926768,1344,44634,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +20891930864,4512,44646,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +20891937008,1120,44659,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +20891941136,1632,44671,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +20891945232,1760,44684,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +20891949296,1280,44695,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20891953360,4640,44716,336,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +20891959536,3591232,44742,37296,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +20895552752,1952,44756,6,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +20895556848,5457248,44767,391608,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20901015792,6822208,44781,783216,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20907840752,45056,44807,414,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +20907887856,6688,44821,6,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +20907896176,51296,44832,4347,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20907950416,61632,44846,8694,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20908014832,4410976,44867,2,583,1,8,16,1,80,0.009,0.000,,,,,NVIDIA GB10 (0),1,,7,"void magma_sgemmEx_kernel(int, int, int, Tensor, int, Tensor, int, Tensor, int, Tensor, int, int, int, const T1 *, const T1 *, T1, T1, int, cublasLtEpilogue_t, int, const void *, long)" +20912428368,99968,44891,13,1,3,128,1,1,80,0.000,0.026,,,,,NVIDIA GB10 (0),1,,7,void cutlass::Kernel2(T1::Params) +20912530672,4000,44894,1,26,1,32,16,1,46,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void cublasLt::splitKreduce_kernel<(int)32, (int)16, int, float, float, float, float, (bool)0, float, float, float, (bool)1, (bool)1, (bool)0, (bool)0>(cublasLt::cublasSplitKParams, const T4 *, const T10 *, T9 *, T5 *, const T6 *, const T6 *, const T11 *, const T4 *, T11 *, void *, long, T6 *, int *, T6 *, T6 *, const T6 *, const T6 *, const T6 *, const T6 *, const T6 *)" +20912536816,59776,44908,3497,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +20912599248,62272,44923,6993,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +20912662832,1280,44938,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +20912666864,100448,44953,13986,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20912769392,86400,44965,3497,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::neg_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +20912857424,1856,44979,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +20912861424,1472,44994,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +20912865616,5600,45006,26,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20912873712,1312,45017,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +20912877904,1536,45028,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +20912882064,1216,45039,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +20912886000,1376,45053,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +20912890320,1664,45065,13,1,1,128,1,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::neg_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 9)]::operator ()() const::[lambda(c10::BFloat16) (instance 1)], std::array>(int, T2, T3)" +20912894288,3360,45076,26,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20912900336,4128,45087,26,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl_nocast>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20912906576,2272,45101,26,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +20912910576,72864,45113,13986,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20912985552,162432,45124,3497,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add, std::array>(int, T2, T3)" +20913149232,2240,45135,52,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20913153264,2016,45146,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add, std::array>(int, T2, T3)" +20913157072,2848,45157,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel::CompareEqFunctor>, std::array, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithoutCast, at::native::memory::StoreWithoutCast>(int, T1, T2, T4, T5, T6, T7)" +20913166480,5664,45163,,,,,,,,,,0.000,0.177,Device,Pinned,NVIDIA GB10 (0),1,,7,[CUDA memcpy Device-to-Host] +20913239056,159936,45174,3497,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add, std::array>(int, T2, T3)" +20913400752,83872,45185,13986,1,1,128,1,1,20,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20913486800,1184,45196,1,1,1,128,1,1,30,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithoutCast, at::native::memory::StoreWithoutCast>(int, T1, T2, T4, T5, T6, T7)" +20913489008,88064,45207,13986,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20913578192,172960,45218,3497,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add, std::array>(int, T2, T3)" +20913753040,1824,45229,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel::CompareEqFunctor>, std::array, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithoutCast, at::native::memory::StoreWithoutCast>(int, T1, T2, T4, T5, T6, T7)" +20913758992,1792,45235,,,,,,,,,,0.000,0.558,Device,Pinned,NVIDIA GB10 (0),1,,7,[CUDA memcpy Device-to-Host] +20913773072,2336,45246,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add, std::array>(int, T2, T3)" +20913782096,3904,45257,52,1,1,128,1,1,20,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20913795024,4928,45268,1,1,1,128,1,1,30,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithoutCast, at::native::memory::StoreWithoutCast>(int, T1, T2, T4, T5, T6, T7)" +20913801104,27520,45279,52,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20913830896,1600,45290,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add, std::array>(int, T2, T3)" +20913858544,1120,45301,13,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" diff --git a/benchmarks/gb10-post-fc2-warmed-step-stats_cuda_kern_exec_trace.csv b/benchmarks/gb10-post-fc2-warmed-step-stats_cuda_kern_exec_trace.csv new file mode 100644 index 0000000..75447d0 --- /dev/null +++ b/benchmarks/gb10-post-fc2-warmed-step-stats_cuda_kern_exec_trace.csv @@ -0,0 +1,2745 @@ +API Start (ns),API Dur (ns),Queue Start (ns),Queue Dur (ns),Kernel Start (ns),Kernel Dur (ns),Total Dur (ns),PID,TID,DevId,API Function,GridXYZ,BlockXYZ,Kernel Name +7267792,13408,,,7275376,2144,13408,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +7290064,5728,,,7292688,2816,5728,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +7303328,5504,,,7305872,3392,5936,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7312624,3072,,,7313232,1088,3072,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +7320240,3056,,,7320496,1088,3056,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +7326912,2624,,,7326704,1056,2624,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +7332464,2944,,,7332624,2080,2944,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7339552,2752,,,7339504,1824,2752,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7354064,6960,,,7357872,2752,6960,73,73,0,cudaLaunchKernel, 13 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +7368368,6496,,,7370736,5728,8096,73,73,0,cudaLaunchKernel, 26 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7381248,2672,,,7381104,1120,2672,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +7392128,4000,,,7392976,1984,4000,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BUnaryFunctor>, std::array>(int, T2, T3)" +7416896,3616,,,7417616,38432,39152,73,73,0,cudaLaunchKernel,3497 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +7457216,5488,,,7459632,48832,51248,73,73,0,cudaLaunchKernel,6993 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7471440,6608,7478048,31952,7510000,71520,110080,73,73,0,cudaLaunchKernel,6993 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +7645392,5456,,,7647600,4915936,4918144,73,73,0,cuLaunchKernel,2336 3 1, 256 1 1,void cutlass::Kernel2(T1::Params) +7660272,3168,7663440,4902048,12565488,5035136,9940352,73,73,0,cudaLaunchKernel,195804 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +7680208,3104,7683312,9918336,17601648,3808,9925248,73,73,0,cudaLaunchKernel, 26 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +7749040,5552,7754592,9853168,17607760,44096,9902816,73,73,0,cuLaunchKernel, 56 6 1, 128 1 1,void cutlass::Kernel2(T1::Params) +7760960,2784,7763744,9890000,17653744,30368,9923152,73,73,0,cudaLaunchKernel,2174 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +7789584,14368,7803952,9881536,17685488,3744,9899648,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +7808960,2624,7811584,9880144,17691728,1792,9884560,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +7816096,3040,7819136,9876944,17696080,1344,9881328,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +7822960,3856,7826816,9873264,17700080,2048,9879168,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7830640,4176,7834816,9869456,17704272,1536,9875168,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +7838288,3408,7841696,9866672,17708368,1280,9871360,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +7846032,2512,7848544,9863824,17712368,1376,9867712,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +7851344,3104,7854448,9862112,17716560,2080,9867296,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7884416,5552,7889968,9830432,17720400,2848,9838832,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnOther_add, std::array>(int, T2, T3)" +17749936,2864,,,17749968,1024,2864,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnOther_add, std::array>(int, T2, T3)" +17804720,6544,,,17807664,3766816,3769760,73,73,0,cudaLaunchKernel,1536 3 1, 128 1 1,"void at::native::::CatArrayBatchedCopy_alignedK_contig::OpaqueType<(unsigned int)2>, unsigned int, (int)2, (int)128, (int)1, (int)8>(T1 *, at::native::::CatArrInputTensorMetadata, at::native::::TensorSizeStride, int, T2)" +26357072,19776,,,26372208,8192,23328,73,73,0,cudaLaunchKernel, 222 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +26411856,6864,,,26413968,27680,29792,73,73,0,cudaLaunchKernel,7090 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +26448080,6096,,,26449776,52384,54080,73,73,0,cudaLaunchKernel, 96 3 1, 512 1 1,"void at::native::::CatArrayBatchedCopy::OpaqueType<(unsigned int)4>, unsigned int, (int)2, (int)64, (int)64>(T1 *, at::native::::CatArrInputTensorMetadata, at::native::::TensorSizeStride, int, T2)" +26467520,8432,26475952,27200,26503152,31872,67504,73,73,0,cudaLaunchKernel,1773 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::cos_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +26485712,7200,26492912,43104,26536016,39456,89760,73,73,0,cudaLaunchKernel,1773 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::sin_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +26503344,7696,26511040,66864,26577904,39712,114272,73,73,0,cudaLaunchKernel,1773 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::neg_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +26524576,5744,26530320,88544,26618864,873696,967984,73,73,0,cudaLaunchKernel,1536 4 1, 128 1 1,"void at::native::::CatArrayBatchedCopy_alignedK_contig::OpaqueType<(unsigned int)4>, unsigned int, (int)3, (int)128, (int)1, (int)16>(T1 *, at::native::::CatArrInputTensorMetadata, at::native::::TensorSizeStride, int, T2)" +26542304,3824,26546128,949280,27495408,217120,1170224,73,73,0,cudaLaunchKernel,7090 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +26580608,5376,26585984,1128560,27714544,2240,1136176,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +26596912,4336,26601248,1117488,27718736,1536,1123360,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +26610752,7728,26618480,1104256,27722736,3424,1115408,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +26623808,10192,26634000,1094848,27728848,3168,1108208,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +26644832,6224,26651056,1082016,27733072,3136,1091376,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +26750464,31104,26781568,955920,27737488,4736,991760,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +26817168,16720,26833888,909616,27743504,1088,927424,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +26842128,5984,26848112,899456,27747568,1632,907072,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +26881584,13776,26895360,856176,27751536,1504,871456,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +26933984,31456,26965440,790320,27755760,2016,823792,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +27338800,37296,27376096,383760,27759856,30048,451104,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +27606992,42304,27649296,143072,27792368,3544672,3730048,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +27744576,4640,27749216,3589264,31338480,2464,3596368,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +27753472,3216,27756688,3585888,31342576,1984,3591088,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +27834368,7616,27841984,3504688,31346672,49568,3561872,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +27856400,16064,27872464,3526432,31398896,3452992,6995488,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +27875376,14016,27889392,6964448,34853840,2912,6981376,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +27895376,18464,27913840,6944128,34857968,2466752,9429344,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +28438432,12560,28450992,8875072,37326064,1920,8889552,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +28955424,19216,28974640,8355520,37330160,24059776,32434512,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +29491888,43168,29535056,31856160,61391216,11842304,43741632,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +29793040,28320,29821360,43430944,73252304,2395136,45854400,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +29899920,23168,29923088,45726176,75649264,4001984,49751328,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +29971760,19104,29990864,49663008,79653872,3999008,53681120,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +30031744,16336,30048080,53606784,83654864,5297792,58920912,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +30084848,13088,30097936,58857152,88955088,5731328,64601568,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +30238096,12496,30250592,64438928,94689520,240315488,304766912,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +30402032,15632,30417664,304589328,335006992,2176224,306781184,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +30424080,12784,30436864,306748144,337185008,1952,306762880,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +30460496,9984,30470480,306718624,337189104,1376,306729984,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +30491344,5984,30497328,306695872,337193200,1504,306703360,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +30616272,10752,30627024,306570272,337197296,40608,306621632,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +30657360,27824,30685184,306555088,337240272,2775488,309358400,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +30757632,9776,30767408,309249984,340017392,1888,309261648,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +30860752,10288,30871040,309150448,340021488,7702240,316862976,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +31210832,13904,31224736,316500304,347725040,5461056,321975264,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +31254736,8192,31262928,321925152,353188080,3717632,325650976,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +31305920,8032,31313952,325593296,356907248,5856,325607184,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +31319328,5072,31324400,325591040,356915440,3776,325599888,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +31353824,3712,31357536,325564048,356921584,60928,325628688,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +31360320,3296,31363616,325620432,356984048,4091840,329715568,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +31364384,3520,31367904,329710064,361077968,1952,329715536,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +31368512,3872,31372384,329709712,361082096,2437024,332150608,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +31411328,3808,31415136,332105296,363520432,1920,332111024,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +31464768,3056,31467824,332056512,363524336,34563520,366623088,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +31511392,3584,31514976,366574512,398089488,140224,366718320,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +31517360,5072,31522432,366709360,398231792,11923520,378637952,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +31522864,2624,31525488,378631808,410157296,2464,378636896,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +31526352,4816,31531168,378630224,410161392,10756480,389391520,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +31555888,2864,31558752,389361808,420920560,1920,389366592,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +31570192,4048,31574240,389350416,420924656,3040,389357504,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +31658144,4288,31662432,389267344,420929776,16202336,405473968,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +31696912,2832,31699744,405433840,437133584,5455232,410891904,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +31715264,4176,31719440,410871008,442590448,1984,410877168,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +31725792,3056,31728848,410865696,442594544,1440,410870192,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +31733712,3488,31737200,410861440,442598640,1376,410866304,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +31742144,3552,31745696,410857040,442602736,3168,410863760,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +31751328,3232,31754560,410854288,442608848,1344,410858864,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +31768288,4032,31772320,410840720,442613040,4608,410849360,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +31778848,3312,31782160,410836832,442618992,1120,410841264,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +31786592,2992,31789584,410833632,442623216,1632,410838256,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +31795744,2992,31798736,410828576,442627312,1472,410833040,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +31806144,3264,31809408,410822000,442631408,1248,410826512,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +31882768,3856,31886624,410748880,442635504,29472,410782208,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +31913904,2912,31916816,410749440,442666256,3479584,414231936,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +31933664,2976,31936640,414212208,446148848,2368,414217552,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +31940592,2832,31943424,414209520,446152944,2112,414214464,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +31960448,2640,31963088,414193952,446157040,48416,414245008,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +31965232,2880,31968112,414240128,446208240,3449888,417692896,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +31968512,2416,31970928,417689216,449660144,4032,417695664,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +31971488,2752,31974240,417692048,449666288,2435424,420130224,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +32013936,2976,32016912,420087520,452104432,1888,420092384,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +32061952,2944,32064896,420043632,452108528,25770432,445817008,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +32179824,4736,32184560,445697024,477881584,11819968,457521728,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +32218688,3456,32222144,457482480,489704624,2346688,459832624,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +32238848,3584,32242432,459811312,492053744,3809824,463624720,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +32246832,2832,32249664,463615408,495865072,3947776,467566016,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +32258208,3104,32261312,467554352,499815664,5406624,472964080,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +32269968,3104,32273072,472951360,505224432,5664096,478618560,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +32286592,3648,32290240,478601008,510891248,240444576,719049232,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +32323632,3488,32327120,719011648,751338768,2172448,721187584,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +32327984,3088,32331072,721181584,753512656,1952,721186624,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +32338128,3184,32341312,721175472,753516784,2912,721181568,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +32346992,3056,32350048,721170928,753520976,1536,721175520,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +32372752,2944,32375696,721149312,753525008,46304,721198560,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +32388960,3136,32392096,721180976,753573072,2749568,723933680,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +32420640,3104,32423744,723900848,756324592,1920,723905872,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +32463184,3280,32466464,723862192,756328656,7743936,731609408,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +32507568,2832,32510400,731563888,764074288,5461696,737028416,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +32524576,3072,32527648,737010608,769538256,3462944,740476624,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +32542416,3168,32545584,740456896,773002480,2304,740462368,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +32549600,2704,32552304,740454272,773006576,1984,740458960,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +32568224,2784,32571008,740439664,773010672,46624,740489072,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +32572784,2864,32575648,740483152,773058800,3966080,744452096,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +32576096,2512,32578608,744449216,777027824,1952,744453680,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +32579200,2528,32581728,744450192,777031920,2729952,747182672,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +32612736,2864,32615600,747148352,779763952,1856,747153072,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +32654352,3184,32657536,747110512,779768048,33135360,780249056,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +32685936,2784,32688720,780216992,812905712,186080,780405856,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +32690304,3744,32694048,780400048,813094096,12695904,793099696,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +32694496,2816,32697312,793094416,825791728,2176,793099408,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +32697712,2880,32700592,793095328,825795920,10475328,803573536,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +32715072,2736,32717808,803554720,836272528,2176,803559632,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +32724256,2976,32727232,803549232,836276464,832,803553040,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +32758416,2864,32761280,803517072,836278352,15614944,819134880,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +32788016,2768,32790784,819103792,851894576,5572000,824678560,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +32803040,3536,32806576,824662592,857469168,2336,824668464,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +32812352,3120,32815472,824657792,857473264,3104,824664016,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +32819408,2592,32822000,824657408,857479408,3264,824663264,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +32826160,2928,32829088,824656464,857485552,11680,824671072,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +32833712,3520,32837232,824662624,857499856,6144,824672288,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +32853184,3360,32856544,824652816,857509360,5600,824661776,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +32861760,3376,32865136,824651136,857516272,2368,824656880,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +32869712,3088,32872800,824647568,857520368,3360,824654016,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +32877088,2624,32879712,824646800,857526512,2944,824652368,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +32885056,3264,32888320,824642416,857530736,8192,824653872,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +32952736,3216,32955952,824584896,857540848,56608,824644720,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +32981504,2896,32984400,824615840,857600240,3673856,828292592,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +32999120,2944,33002064,828273312,861275376,2464,828278720,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +33006048,2736,33008784,828270656,861279440,2048,828275440,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +33024304,2640,33026944,828256592,861283536,48960,828308192,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +33028848,2640,33031488,828303248,861334736,3813056,832118944,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +33031888,2800,33034688,832115472,865150160,3776,832122048,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +33035216,2496,33037712,832118592,865156304,2434944,834556032,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +33080784,2976,33083760,834509696,867593456,1888,834514560,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +33125728,3296,33129024,834468560,867597584,23937632,858409488,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +33213920,4080,33218000,858319648,891537648,12480864,870804592,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +33245472,3360,33248832,870772624,904021456,2345600,873121584,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +33263808,3408,33267216,873102048,906369264,3709440,876814896,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +33270080,2816,33272896,876807312,910080208,3724512,880534640,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +33280032,2736,33282768,880524800,913807568,5534656,886062192,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +33290304,2768,33293072,886050432,919343504,5795488,891848688,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +33303040,3056,33306096,891835136,925141232,239349632,1131187824,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +33336304,3424,33339728,1131153472,1164493200,2170784,1133327680,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +33340192,4048,33344240,1133321696,1166665936,1920,1133327664,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +33350944,2816,33353760,1133316304,1166670064,2912,1133322032,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +33358672,3152,33361824,1133312464,1166674288,1504,1133317120,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +33381392,2912,33384304,1133293984,1166678288,47104,1133344000,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +33396976,3360,33400336,1133328064,1166728400,2757664,1136089088,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +33428064,2976,33431040,1136058096,1169489136,1952,1136063024,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +33468256,3136,33471392,1136021840,1169493232,7799744,1143824720,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +33509632,3280,33512912,1143782208,1177295120,5451936,1149237424,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +33523600,3088,33526688,1149222192,1182748880,3475680,1152700960,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +33539920,3248,33543168,1152684272,1186227440,2400,1152689920,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +33547248,3136,33550384,1152681120,1186231504,2080,1152686336,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +33565648,2736,33568384,1152667216,1186235600,46016,1152715968,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +33570064,2768,33572832,1152710064,1186282896,3462240,1156175072,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +33573280,2640,33575920,1156171008,1189746928,1984,1156175632,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +33576480,2656,33579136,1156171856,1189750992,2452064,1158626576,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +33616560,2800,33619360,1158585168,1192204528,1888,1158589856,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +33656784,2640,33659424,1158549200,1192208624,34645696,1193197536,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +33687520,2704,33690224,1193166560,1226856784,139968,1193309232,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +33691632,3184,33694816,1193304208,1226999024,12761440,1206068832,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +33695344,2784,33698128,1206065056,1239763184,1920,1206069760,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +33698672,3184,33701856,1206065424,1239767280,10456096,1216524704,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +33715008,2704,33717712,1216507680,1250225392,1952,1216512336,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +33724112,2832,33726944,1216502512,1250229456,864,1216506208,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +33752400,2656,33755056,1216476576,1250231632,15792128,1232271360,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +33780528,2448,33782976,1232243792,1266026768,5824160,1238070400,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +33794848,3312,33798160,1238054112,1271852272,1952,1238059376,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +33803584,2864,33806448,1238049920,1271856368,1408,1238054192,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +33810304,2768,33813072,1238047360,1271860432,2848,1238052976,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +33817328,3200,33820528,1238044032,1271864560,3200,1238050432,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +33824592,3104,33827696,1238043008,1271870704,1376,1238047488,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +33836384,3024,33839408,1238035392,1271874800,4416,1238042832,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +33844304,2688,33846992,1238033920,1271880912,1120,1238037728,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +33851264,2640,33853904,1238031136,1271885040,1600,1238035376,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +33858224,2944,33861168,1238027968,1271889136,1472,1238032384,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +33866544,3312,33869856,1238023344,1271893200,1280,1238027936,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +33931712,2992,33934704,1237962688,1271897392,26016,1237991696,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +33958256,2912,33961168,1237964800,1271925968,3545984,1241513696,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +33976368,2912,33979280,1241493952,1275473232,2496,1241499360,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +33983168,2576,33985744,1241491488,1275477232,2112,1241496176,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +34001264,3040,34004304,1241477024,1275481328,47552,1241527616,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +34006352,2992,34009344,1241521424,1275530768,4259520,1245783936,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +34009760,2672,34012432,1245780992,1279793424,3712,1245787376,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +34012944,2608,34015552,1245783984,1279799536,2444704,1248231296,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +34050960,3472,34054432,1248191440,1282245872,1824,1248196736,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +34103904,2752,34106656,1248143280,1282249936,24882560,1273028592,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +34187600,3696,34191296,1272943952,1307135248,12475168,1285422816,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +34216080,3328,34219408,1285393248,1319612656,2339296,1287735872,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +34233888,2992,34236880,1287717664,1321954544,3722080,1291442736,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +34239456,2528,34241984,1291436848,1325678832,3736032,1295175408,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +34248800,2864,34251664,1295164768,1329416432,5150368,1300318000,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +34259200,2592,34261792,1300307408,1334569200,5873152,1306183152,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +34270608,2880,34273488,1306170400,1340443888,238904576,1545077856,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +34302512,3696,34306208,1545045104,1579351312,2158912,1547207712,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +34306752,2560,34309312,1547203600,1581512912,1984,1547208144,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +34315600,2880,34318480,1547198560,1581517040,2880,1547204320,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +34323536,9296,34332832,1547188368,1581521200,1536,1547199200,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +34351312,2768,34354080,1547171056,1581525136,54208,1547228032,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +34366992,3264,34370256,1547210368,1581580624,2747456,1549961088,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +34396800,2800,34399600,1549930336,1584329936,1920,1549935056,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +34436816,2784,34439600,1549894464,1584334064,7804864,1557702112,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +34475856,2848,34478704,1557662368,1592141072,5612512,1563277728,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +34489264,3056,34492320,1563263312,1597755632,3661824,1566928192,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +34505120,3200,34508320,1566911184,1601419504,2400,1566916784,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +34512384,2848,34515232,1566908336,1601423568,2112,1566913296,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +34529744,2720,34532464,1566895232,1601427696,46240,1566944192,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +34534256,3040,34537296,1566938528,1601475824,3859104,1570800672,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +34537760,2592,34540352,1570796976,1605337328,1952,1570801520,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +34540848,2416,34543264,1570798192,1605341456,2438848,1573239456,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +34571760,2848,34574608,1573208032,1607782640,1920,1573212800,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +34610608,2720,34613328,1573173408,1607786736,34043744,1607219872,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +34641248,2688,34643936,1607187888,1641831824,139840,1607330416,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +34645392,2848,34648240,1607324704,1641972944,12366720,1619694272,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +34648656,2384,34651040,1619690832,1654341872,1920,1619695136,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +34651472,2992,34654464,1619691504,1654345968,10667840,1630362336,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +34667152,2928,34670080,1630345968,1665016048,1984,1630350880,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +34676304,3024,34679328,1630340816,1665020144,2560,1630346400,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +34702592,2608,34705200,1630319040,1665024240,15704512,1646026160,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +34729936,2912,34732848,1645997504,1680730352,5454720,1651455136,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +34743968,3360,34747328,1651439152,1686186480,1952,1651444464,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +34752832,2992,34755824,1651433920,1686189744,1440,1651438352,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +34759696,2672,34762368,1651431024,1686193392,1312,1651435008,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +34766368,3120,34769488,1651428000,1686197488,3264,1651434384,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +34773616,2736,34776352,1651427248,1686203600,1344,1651431328,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +34784832,3008,34787840,1651419888,1686207728,4640,1651427536,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +34792688,2544,34795232,1651418640,1686213872,1088,1651422272,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +34799952,2784,34802736,1651415232,1686217968,1632,1651419648,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +34806848,2704,34809552,1651412608,1686222160,1504,1651416816,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +34814544,2896,34817440,1651408720,1686226160,1280,1651412896,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +34878832,3136,34881968,1651348288,1686230256,27264,1651378688,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +34907072,3184,34910256,1651348672,1686258928,3493056,1654844912,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +34928800,3280,34932080,1654821792,1689753872,2464,1654827536,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +34936064,2816,34938880,1654819056,1689757936,2016,1654823888,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +34954144,2720,34956864,1654805168,1689762032,47456,1654855344,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +34958784,2800,34961584,1654849568,1689811152,3918208,1658770576,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +34962064,2448,34964512,1658766544,1693731056,3392,1658772384,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +34964976,2544,34967520,1658769680,1693737200,2845728,1661617952,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +35000496,3072,35003568,1661582400,1696585968,1856,1661587328,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +35048800,6928,35055728,1661534368,1696590096,25301536,1686842832,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +35154928,5136,35160064,1686734032,1721894096,11901664,1698640832,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +35184208,5216,35189424,1698608960,1733798384,2551328,1701165504,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +35203680,2864,35206544,1701145440,1736351984,3738560,1704886864,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +35208960,2960,35211920,1704880704,1740092624,3951712,1708835376,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +35218672,4384,35223056,1708823264,1744046320,5119744,1713947392,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +35230784,5216,35236000,1713931344,1749167344,5697824,1719634384,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +35244976,3472,35248448,1719619472,1754867920,240193088,1959816032,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +35284608,3456,35288064,1959774480,1995062544,2167328,1961945264,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +35288640,2688,35291328,1961941040,1997232368,1920,1961945648,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +35298272,3152,35301424,1961935008,1997236432,2944,1961941104,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +35306416,4000,35310416,1961930240,1997240656,1504,1961935744,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +35329728,3936,35333664,1961910960,1997244624,47296,1961962192,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +35346480,3856,35350336,1961943472,1997293808,2750016,1964697344,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +35377088,3360,35380448,1964665840,2000046288,1952,1964671152,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +35421120,4240,35425360,1964625056,2000050416,7806304,1972435600,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +35465136,4736,35469872,1972389600,2007859472,5472288,1977866624,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +35480464,3872,35484336,1977850432,2013334768,3708736,1981563040,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +35498128,3440,35501568,1981544144,2017045712,6624,1981554208,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +35505536,3952,35509488,1981544448,2017053936,2144,1981550544,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +35524032,3344,35527376,1981530624,2017058000,47936,1981581904,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +35529584,3696,35533280,1981573936,2017107216,4063680,1985641312,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +35533744,4272,35538016,1985635472,2021173488,1952,1985641696,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +35538640,4144,35542784,1985634800,2021177584,2438432,1988077376,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +35575616,3312,35578928,1988038848,2023617776,1920,1988044080,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +35615008,3456,35618464,1988003408,2023621872,34049568,2022056432,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +35655808,3200,35659008,2022014960,2057673968,139872,2022158032,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +35660464,3520,35663984,2022152288,2057816272,11888096,2034043904,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +35664912,6000,35670912,2034035024,2069705936,1984,2034043008,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +35671344,3808,35675152,2034034912,2069710064,10741088,2044779808,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +35689952,3264,35693216,2044760656,2080453872,1952,2044765872,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +35698960,3104,35702064,2044755904,2080457968,2496,2044761504,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +35726064,3472,35729536,2044732528,2080462064,16236448,2060972448,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +35754320,3344,35757664,2060942960,2096700624,5467584,2066413888,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +35771184,4720,35775904,2066394928,2102170832,1952,2066401600,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +35782272,3152,35785424,2066389536,2102174960,1440,2066394128,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +35789024,2768,35791792,2066387264,2102179056,1312,2066391344,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +35795968,2832,35798800,2066383104,2102181904,3456,2066389392,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +35803104,3072,35806176,2066381072,2102187248,1344,2066385488,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +35815392,3456,35818848,2066372432,2102191280,4608,2066380496,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +35823776,2896,35826672,2066370816,2102197488,1088,2066374800,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +35831184,2976,35834160,2066367360,2102201520,1632,2066371968,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +35838464,11024,35849488,2066356192,2102205680,1728,2066368944,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +35856272,3136,35859408,2066350368,2102209776,1280,2066354784,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +35920992,3344,35924336,2066289504,2102213840,26368,2066319216,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +35950704,3280,35953984,2066288560,2102242544,3495232,2069787072,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +35970352,3616,35973968,2069766560,2105740528,2496,2069772672,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +35978512,5664,35984176,2069760448,2105744624,1984,2069768096,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +35999808,3184,36002992,2069745728,2105748720,47232,2069796144,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +36005040,3312,36008352,2069789488,2105797840,3515456,2073308256,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +36009264,4784,36014048,2073302256,2109316304,3360,2073310400,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +36015072,4448,36019520,2073302960,2109322480,2434016,2075741424,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +36053872,3184,36057056,2075701520,2111758576,1888,2075706592,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +36107056,3776,36110832,2075651840,2111762672,25849184,2101504800,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +36190640,5216,36195856,2101418752,2137614608,11818912,2113242880,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +36218672,3776,36222448,2113214272,2149436720,2354816,2115572864,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +36235920,3504,36239424,2115553488,2151792912,3750336,2119307328,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +36241808,3184,36244992,2119299792,2155544784,4064384,2123367360,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +36251648,4896,36256544,2123355568,2159612112,5416480,2128776944,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +36264336,3504,36267840,2128763312,2165031152,5667744,2134434560,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +36276400,3280,36279680,2134420496,2170700176,242080160,2376503936,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +36308848,4816,36313664,2376469200,2412782864,2159488,2378633504,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +36314336,4656,36318992,2378625504,2414944496,1952,2378632112,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +36328336,3696,36332032,2378616560,2414948592,2944,2378623200,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +36337120,3536,36340656,2378611904,2414952560,1504,2378616944,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +36358736,4160,36362896,2378592416,2414955312,49536,2378646112,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +36375328,3776,36379104,2378627824,2415006928,2752224,2381383824,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +36407392,4912,36412304,2381348192,2417760496,1888,2381354992,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +36451232,3680,36454912,2381309648,2417764560,7814080,2389127408,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +36492672,3088,36495760,2389085056,2425580816,5456192,2394544336,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +36507072,18896,36525968,2394512704,2431038672,3483520,2398015120,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +36542272,3568,36545840,2397978784,2434524624,6336,2397988688,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +36549792,3424,36553216,2397979376,2434532592,36384,2398019184,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +36567776,3184,36570960,2398000544,2434571504,54208,2398057936,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +36572704,3856,36576560,2398052288,2434628848,4283872,2402340016,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +36577008,3680,36580688,2402333568,2438914256,1952,2402339200,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +36581296,2848,36584144,2402334240,2438918384,2440160,2404777248,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +36620784,6128,36626912,2404733712,2441360624,1888,2404741728,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +36668480,3504,36671984,2404692736,2441364720,33763776,2438460016,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +36703360,3552,36706912,2438424336,2475131248,139488,2438567376,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +36708528,3024,36711552,2438560880,2475272432,12292416,2450856320,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +36712096,2912,36715008,2450852560,2487567568,2176,2450857648,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +36715456,3760,36719216,2450852512,2487571728,10786304,2461642576,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +36733264,3184,36736448,2461624112,2498360560,1920,2461629216,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +36742272,8800,36751072,2461613584,2498364656,832,2461623216,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +36777024,3056,36780080,2461586688,2498366768,15820672,2477410416,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +36803984,3600,36807584,2477382992,2514190576,5490432,2482877024,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +36820768,3200,36823968,2482859376,2519683344,2272,2482864848,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +36829504,4096,36833600,2482853840,2519687440,1408,2482859344,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +36837408,3936,36841344,2482850160,2519691504,3712,2482857808,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +36845664,3024,36848688,2482848960,2519697648,8064,2482860048,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +36852736,4016,36856752,2482851104,2519707856,1568,2482856688,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +36865424,3648,36869072,2482842848,2519711920,8096,2482854592,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +36873712,2848,36876560,2482845664,2519722224,6880,2482855392,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +36880880,3072,36883952,2482846432,2519730384,1632,2482851136,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +36888176,3216,36891392,2482843120,2519734512,14048,2482860384,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +36896080,3040,36899120,2482851776,2519750896,3648,2482858464,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +36960640,3008,36963648,2482793392,2519757040,93376,2482889776,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +36987088,3056,36990144,2482863120,2519853264,3592384,2486458560,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +37012352,4400,37016752,2486431808,2523448560,2912,2486439120,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +37020624,3760,37024384,2486428336,2523452720,2240,2486434336,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +37039008,4400,37043408,2486413344,2523456752,47712,2486465456,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +37045504,3424,37048928,2486456976,2523505904,3622240,2490082640,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +37049360,2816,37052176,2490077632,2527129808,4000,2490084448,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +37052672,2880,37055552,2490080432,2527135984,2437120,2492520432,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +37098880,4256,37103136,2492473072,2529576208,1856,2492479184,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +37145296,3792,37149088,2492431184,2529580272,25405312,2517840288,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +37232240,4576,37236816,2517750048,2554986864,12071296,2529825920,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +37259696,4160,37263856,2529798144,2567062000,2347552,2532149856,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +37277456,3328,37280784,2532130048,2569410832,3711616,2535844992,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +37283104,2512,37285616,2535838208,2573123824,4098912,2539939632,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +37292432,3168,37295600,2539928384,2577223984,5356480,2545288032,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +37302816,2704,37305520,2545276992,2582582512,5691456,2550971152,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +37313312,3648,37316960,2550960048,2588277008,239531264,2790494960,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +37344224,3120,37347344,2790462720,2827810064,2158528,2792624368,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +37347904,2576,37350480,2792620160,2829970640,1984,2792624720,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +37356720,3328,37360048,2792614720,2829974768,2720,2792620768,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +37364784,4608,37369392,2792609472,2829978864,1504,2792615584,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +37393408,2928,37396336,2792586592,2829982928,46080,2792635600,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +37409136,3120,37412256,2792619824,2830032080,2749120,2795372064,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +37439296,2944,37442240,2795341360,2832783600,1920,2795346224,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +37479072,2704,37481776,2795305920,2832787696,7819616,2803128240,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +37516160,3232,37519392,2803090640,2840610032,5449728,2808543600,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +37532224,3216,37535440,2808527392,2846062832,3479168,2812009776,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +37549472,3072,37552544,2811990864,2849543408,2336,2811996272,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +37556528,3008,37559536,2811987968,2849547504,1984,2811992960,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +37573568,3168,37576736,2811974864,2849551600,46464,2812024496,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +37578480,3008,37581488,2812019232,2849600720,3413024,2815435264,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +37581936,2864,37584800,2815430992,2853015792,1984,2815435840,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +37585360,2768,37588128,2815431760,2853019888,2757152,2818191680,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +37617264,4032,37621296,2818158240,2855779536,1856,2818164128,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +37659664,3504,37663168,2818120464,2855783632,33791744,2851915712,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +37691024,2912,37693936,2851882784,2889576720,141536,2852027232,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +37695328,2832,37698160,2852022912,2889721072,12811136,2864836880,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +37698544,2608,37701152,2864833200,2902534352,2208,2864838016,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +37701600,2752,37704352,2864834128,2902538480,10362432,2875199312,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +37716720,2576,37719296,2875184112,2912903408,1952,2875188640,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +37724768,2944,37727712,2875179760,2912907472,864,2875183568,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +37749280,2768,37752048,2875157568,2912909616,15744352,2890904688,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +37775392,3920,37779312,2890877312,2928656624,5465664,2896346896,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +37791264,3024,37794288,2896330464,2934124752,1984,2896335472,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +37799824,3520,37803344,2896325568,2934128912,1408,2896330496,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +37807136,3280,37810416,2896322560,2934132976,3680,2896329520,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +37814592,3632,37818224,2896320896,2934139120,3488,2896328016,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +37822464,3120,37825584,2896319552,2934145136,1344,2896324016,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +37834608,3552,37838160,2896309568,2934147728,4608,2896317728,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +37842928,3200,37846128,2896308224,2934154352,1120,2896312544,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +37850432,3408,37853840,2896304736,2934158576,1632,2896309776,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +37858000,3136,37861136,2896301536,2934162672,1472,2896306144,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +37866208,3088,37869296,2896297472,2934166768,1280,2896301840,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +37929552,2752,37932304,2896238560,2934170864,25920,2896267232,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +37955648,3056,37958704,2896240832,2934199536,3495488,2899739376,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +37973808,2992,37976800,2899719696,2937696496,2528,2899725216,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +37980720,2896,37983616,2899716944,2937700560,1920,2899721760,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +37998832,2784,38001616,2899703040,2937704656,48864,2899754688,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +38003584,2944,38006528,2899748336,2937754864,3453632,2903204912,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +38006944,2448,38009392,2903201440,2941210832,3808,2903207696,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +38009872,2736,38012608,2903204400,2941217008,2435392,2905642528,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +38055008,2928,38057936,2905596224,2943654160,1888,2905601040,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +38108384,2928,38111312,2905546912,2943658224,24871360,2930421200,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +38195520,3904,38199424,2930332752,2968532176,12473728,2942810384,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +38223248,4864,38228112,2942781504,2981009616,2342880,2945129248,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +38242272,2864,38245136,2945109472,2983354608,3711520,2948823856,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +38247584,2784,38250368,2948817264,2987067632,3730304,2952550352,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +38257152,2576,38259728,2952540384,2990800112,5451680,2957994640,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +38267296,2592,38269888,2957983504,2996253392,5726816,2963712912,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +38277792,2768,38280560,2963702656,3001983216,240209920,3203915344,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +38309616,3088,38312704,3203883536,3242196240,2180576,3206067200,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +38313152,2448,38315600,3206063776,3244379376,1984,3206068208,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +38322912,3104,38326016,3206057488,3244383504,2752,3206063344,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +38330832,2832,38333664,3206053872,3244387536,1504,3206058208,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +38351760,2896,38354656,3206037008,3244391664,45632,3206085536,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +38367104,3072,38370176,3206068560,3244438736,2899360,3208970992,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +38396480,3024,38399504,3208941280,3247340784,2016,3208946320,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +38437408,2688,38440096,3208904752,3247344848,8260608,3217168048,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +38475872,2560,38478432,3217130128,3255608560,5606272,3222738960,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +38489328,2960,38492288,3222724752,3261217040,3532864,3226260576,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +38505936,3264,38509200,3226242624,3264751824,2528,3226248416,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +38513008,2928,38515936,3226239984,3264755920,1984,3226244896,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +38530208,3168,38533376,3226226672,3264760048,46304,3226276144,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +38535184,3664,38538848,3226269296,3264808144,3517760,3229790720,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +38539344,3440,38542784,3229784880,3268327664,2144,3229790464,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +38543360,2480,38545840,3229785920,3268331760,2417696,3232206096,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +38574352,2848,38577200,3232174272,3270751472,1888,3232179008,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +38613264,2800,38616064,3232139504,3270755568,34255616,3266397920,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +38645744,3136,38648880,3266364672,3305013552,141536,3266509344,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +38650320,3040,38653360,3266504512,3305157872,12819968,3279327520,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +38653936,2960,38656896,3279322448,3317979344,2464,3279327872,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +38657648,3616,38661264,3279322208,3317983472,10346848,3289672672,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +38675312,3488,38678800,3289654240,3328333040,2144,3289659872,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +38684480,4000,38688480,3289648656,3328337136,864,3289653520,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +38711056,3552,38714608,3289624672,3328339280,15735488,3305363712,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +38737776,2624,38740400,3305336640,3344077040,5454016,3310793280,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +38752544,3856,38756400,3310776512,3349532912,2016,3310782384,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +38764400,3424,38767824,3310769184,3349537008,1408,3310774016,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +38771568,3488,38775056,3310766048,3349541104,1312,3310770848,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +38778976,3232,38782208,3310762992,3349545200,3456,3310769680,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +38786480,3280,38789760,3310761584,3349551344,1312,3310766176,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +38798528,3984,38802512,3310753056,3349555568,4224,3310761264,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +38807280,2832,38810112,3310751408,3349561520,1088,3310755328,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +38814272,2928,38817200,3310748480,3349565680,1632,3310753040,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +38821696,3744,38825440,3310744304,3349569744,1504,3310749552,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +38830464,3072,38833536,3310740336,3349573872,1280,3310744688,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +38913568,3008,38916576,3310661360,3349577936,26720,3310691088,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +38942128,3168,38945296,3310661344,3349606640,3496640,3314161152,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +38965424,3504,38968928,3314135824,3353104752,2400,3314141728,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +38972752,2912,38975664,3314133056,3353108720,2112,3314138080,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +38991056,2992,38994048,3314118768,3353112816,47680,3314169440,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +38996160,3120,38999280,3314163680,3353162960,3506560,3317673360,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +38999728,2464,39002192,3317670016,3356672208,3680,3317676160,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +39002720,2512,39005232,3317673120,3356678352,2438752,3320114384,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +39040000,2816,39042816,3320076784,3359119600,1888,3320081488,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +39094368,2976,39097344,3320026320,3359123664,24909184,3344938480,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +39177216,3904,39181120,3344854448,3384035568,12198048,3357056400,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +39203856,3760,39207616,3357029168,3396236784,2509216,3359542144,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +39221088,3152,39224240,3359523232,3398747472,3777248,3363303632,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +39226560,2752,39229312,3363296688,3402526000,3770368,3367069808,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +39236096,2992,39239088,3367060256,3406299344,5082752,3372146000,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +39246496,2816,39249312,3372135248,3411384560,5710464,3377848528,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +39256992,3488,39260480,3377837424,3417097904,240101408,3617942320,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +39290672,3024,39293696,3617907216,3657200912,2166688,3620076928,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +39294160,2768,39296928,3620073808,3659370736,1920,3620078496,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +39303136,3152,39306288,3620068544,3659374832,2528,3620074224,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +39310944,3008,39313952,3620064944,3659378896,1536,3620069488,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +39330736,2976,39333712,3620049184,3659382896,45856,3620098016,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +39345776,3440,39349216,3620081904,3659431120,2762624,3622847968,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +39375168,5648,39380816,3622815104,3662195920,1920,3622822672,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +39418000,2976,39420976,3622779040,3662200016,8099264,3630881280,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +39456784,2592,39459376,3630841600,3670300976,5523968,3636368160,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +39469776,3200,39472976,3636353472,3675826448,3822336,3640179008,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +39486640,3344,39489984,3640160080,3679650064,2528,3640165952,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +39493872,3344,39497216,3640156912,3679654128,2080,3640162336,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +39511136,3280,39514416,3640143808,3679658224,46624,3640193712,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +39516128,3728,39519856,3640187488,3679707344,3806144,3643997360,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +39520288,3552,39523840,3643990928,3683514768,1920,3643996400,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +39524400,4752,39529152,3643989552,3683518704,2444096,3646438400,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +39558256,2864,39561120,3646403920,3685965040,1920,3646408704,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +39597840,3616,39601456,3646367680,3685969136,34983616,3681354912,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +39630896,3184,39634080,3681320080,3720954160,139744,3681463008,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +39635648,3440,39639088,3681456320,3721095408,12296512,3693756272,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +39639504,2592,39642096,3693752544,3733394640,9696,3693764832,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +39642576,2624,39645200,3693761760,3733406960,10774656,3704539040,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +39667072,3872,39670944,3704513616,3744184560,7232,3704524720,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +39676736,3872,39680608,3704514320,3744194928,3328,3704521520,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +39703200,2864,39706064,3704494912,3744200976,15784672,3720282448,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +39729776,2608,39732384,3720254608,3759986992,5478976,3725736192,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +39744400,3600,39748000,3725720400,3765468400,1952,3725725952,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +39753696,3280,39756976,3725715520,3765472496,1440,3725720240,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +39760768,3296,39764064,3725712528,3765476592,1280,3725717104,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +39767872,3264,39771136,3725709520,3765480656,3456,3725716240,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +39775184,3504,39778688,3725708112,3765486800,1312,3725712928,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +39787168,4032,39791200,3725699600,3765490800,4448,3725708080,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +39795856,3600,39799456,3725697584,3765497040,1312,3725702496,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +39803984,3072,39807056,3725692672,3765499728,1632,3725697376,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +39811120,3008,39814128,3725689088,3765503216,1472,3725693568,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +39818976,3296,39822272,3725685040,3765507312,1280,3725689616,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +39883376,3584,39886960,3725624448,3765511408,24768,3725652800,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +39910752,3136,39913888,3725624144,3765538032,3497504,3729124784,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +39929792,3168,39932960,3729104080,3769037040,2464,3729109712,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +39936864,3872,39940736,3729100400,3769041136,2016,3729106288,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +39955568,3680,39959248,3729085984,3769045232,48544,3729138208,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +39961248,3008,39964256,3729131120,3769095376,3424864,3732558992,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +39964912,3120,39968032,3732554736,3772522768,3552,3732561408,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +39968656,3120,39971776,3732557136,3772528912,2510688,3735070944,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +40005600,3392,40008992,3735032784,3775041776,1888,3735038064,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +40050128,3184,40053312,3734992560,3775045872,24845056,3759840800,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +40142992,3632,40146624,3759746608,3799893232,11830912,3771581152,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +40169056,3440,40172496,3771553984,3811726480,2388928,3773946352,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +40185728,3184,40188912,3773928704,3814117616,3942304,3777874192,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +40191168,3056,40194224,3777867808,3818062032,3950304,3781821168,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +40201120,3424,40204544,3781810192,3822014736,5234368,3787047984,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +40211856,3824,40215680,3787034736,3827250416,5660672,3792699232,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +40223504,3312,40226816,3792686320,3832913136,240325056,4033014688,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +40257104,3344,40260448,4032980400,4073240848,2157632,4035141376,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +40260928,2688,40263616,4035136784,4075400400,1952,4035141424,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +40270096,3296,40273392,4035131104,4075404496,2560,4035136960,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +40278304,3280,40281584,4035127040,4075408624,1504,4035131824,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +40298560,2992,40301552,4035111232,4075412784,50400,4035164624,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +40313952,3136,40317088,4035148848,4075465936,2760608,4037912592,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +40342080,3616,40345696,4037882160,4078227856,1952,4037887728,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +40382288,2880,40385168,4037846624,4078231792,7826784,4045676288,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +40427632,2976,40430608,4045630720,4086061328,5463200,4051096896,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +40440912,3392,40444304,4051082048,4091526352,3570432,4054655872,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +40458800,3744,40462544,4054636640,4095099184,2336,4054642720,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +40466544,3232,40469776,4054633408,4095103184,2208,4054638848,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +40484256,3184,40487440,4054619872,4095107312,48384,4054671440,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +40489216,3200,40492416,4054665072,4095157488,4305856,4058974128,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +40492944,2896,40495840,4058969616,4099465456,1920,4058974432,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +40496384,3168,40499552,4058970032,4099469584,2451072,4061424272,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +40530736,3024,40533760,4061389328,4101923088,1856,4061394208,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +40570688,3232,40573920,4061353200,4101927120,33911744,4095268176,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +40602272,4112,40606384,4095234624,4135841008,206816,4095445552,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +40607856,2976,40610832,4095440096,4136050928,12091680,4107534752,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +40611248,2736,40613984,4107530352,4148144336,2208,4107535296,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +40614416,2848,40617264,4107531200,4148148464,10751008,4118285056,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +40629648,2832,40632480,4118270032,4158902512,1920,4118274784,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +40637920,2672,40640592,4118266016,4158906608,3328,4118272016,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +40663168,3264,40666432,4118246320,4158912752,15259232,4133508816,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +40689488,4000,40693488,4133480160,4174173648,5466400,4138950560,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +40706640,4048,40710688,4138931984,4179642672,1920,4138937952,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +40715904,3392,40719296,4138927408,4179646704,1440,4138932240,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +40723024,2944,40725968,4138924800,4179650768,1312,4138929056,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +40729808,3264,40733072,4138921824,4179654896,3072,4138928160,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +40737152,3280,40740432,4138920608,4179661040,1312,4138925200,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +40749472,3264,40752736,4138912304,4179665040,4800,4138920368,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +40757424,2800,40760224,4138911024,4179671248,1120,4138914944,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +40764736,2992,40767728,4138907584,4179675312,1632,4138912208,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +40771824,2960,40774784,4138904688,4179679472,1568,4138909216,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +40779488,3328,40782816,4138900720,4179683536,1248,4138905296,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +40844480,3120,40847600,4138840096,4179687696,28672,4138871888,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +40869664,3360,40873024,4138845360,4179718384,3498816,4142347536,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +40890320,3984,40894304,4142324176,4183218480,2432,4142330592,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +40898320,3280,40901600,4142320912,4183222512,2112,4142326304,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +40917184,3712,40920896,4142305680,4183226576,48096,4142357488,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +40923136,3152,40926288,4142350464,4183276752,3518400,4145872016,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +40926736,2864,40929600,4145866864,4186796464,3456,4145873184,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +40930064,2896,40932960,4145869488,4186802448,2435904,4148308288,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +40967040,3232,40970272,4148270288,4189240560,1888,4148275408,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +41011104,2800,41013904,4148230752,4189244656,26090176,4174323728,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +41102608,4128,41106736,4174230464,4215337200,12096896,4186331488,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +41130176,3280,41133456,4186303552,4227437008,2347104,4188653936,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +41146768,3696,41150464,4188636400,4229786864,3730976,4192371072,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +41152720,3984,41156704,4192362608,4233519312,4067328,4196433920,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +41168160,3184,41171344,4196417344,4237588688,5428096,4201848624,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +41178784,2928,41181712,4201837312,4243019024,5745408,4207585648,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +41189504,2800,41192304,4207574400,4248766704,242399936,4449977136,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +41222176,3840,41226016,4449942032,4491168048,2168800,4452114672,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +41226608,2656,41229264,4452109632,4493338896,1920,4452114208,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +41235472,3184,41238656,4452104304,4493342960,2880,4452110368,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +41243360,2976,41246336,4452100848,4493347184,1504,4452105328,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +41263632,5136,41268768,4452082416,4493351184,46592,4452134144,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +41281440,3328,41284768,4452114480,4493399248,2748320,4454866128,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +41311552,3232,41314784,4454834960,4496149744,1952,4454840144,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +41351648,3280,41354928,4454798944,4496153872,7929856,4462732080,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +41396240,3328,41399568,4462687008,4504086576,5458016,4468148352,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +41410160,3456,41413616,4468133120,4509546736,3492480,4471629056,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +41428480,3344,41431824,4471609920,4513041744,2400,4471615664,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +41435856,3264,41439120,4471606592,4513045712,1984,4471611840,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +41453632,3040,41456672,4471593168,4513049840,47232,4471643440,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +41458464,2768,41461232,4471638784,4513100016,4172256,4475813808,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +41461792,2768,41464560,4475809248,4517273808,1984,4475814000,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +41465120,3056,41468176,4475809760,4517277936,2537024,4478349840,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +41497568,5216,41502784,4478313648,4519816432,1888,4478320752,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +41538912,2960,41541872,4478278688,4519820560,33533088,4511814736,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +41568368,3408,41571776,4511783760,4553355536,157600,4511944768,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +41573200,3056,41576256,4511938992,4553515248,12645408,4524587456,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +41576672,2544,41579216,4524583424,4566162640,2432,4524588400,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +41579616,2624,41582240,4524584528,4566166768,10464160,4535051312,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +41597904,2864,41600768,4535031440,4576632208,2400,4535036704,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +41606352,3200,41609552,4535026592,4576636144,2208,4535032000,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +41631648,5152,41636800,4535003440,4576640240,15692288,4550700880,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +41660768,2880,41663648,4550670416,4592334064,5631424,4556304720,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +41675808,3024,41678832,4556289280,4597968112,2560,4556294864,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +41684272,3296,41687568,4556284640,4597972208,1440,4556289376,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +41691328,2960,41694288,4556281984,4597976272,1472,4556286416,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +41699664,3056,41702720,4556277648,4597980368,3680,4556284384,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +41706768,2784,41709552,4556276928,4597986480,3136,4556282848,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +41718288,3824,41722112,4556270576,4597992688,6432,4556280832,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +41727056,2976,41730032,4556270848,4598000880,1280,4556275104,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +41734256,3424,41737680,4556267296,4598004976,7072,4556277792,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +41741776,2768,41744544,4556268880,4598013424,11168,4556282816,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +41749424,3360,41752784,4556274720,4598027504,1248,4556279328,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +41813760,3216,41816976,4556214624,4598031600,71488,4556289328,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +41839776,3104,41842880,4556262416,4598105296,3607424,4559872944,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +41857984,2992,41860976,4559853024,4601714000,2464,4559858480,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +41864976,3152,41868128,4559849872,4601718000,2080,4559855104,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +41883168,2768,41885936,4559836160,4601722096,48672,4559887600,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +41887840,3056,41890896,4559881408,4601772304,3869632,4563754096,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +41891296,3440,41894736,4563750272,4605645008,4032,4563757744,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +41895280,2944,41898224,4563752928,4605651152,2434688,4566190560,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +41940368,4848,41945216,4566142064,4608087280,1856,4566148768,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +41986240,3168,41989408,4566102000,4608091408,25006880,4591112048,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +42064880,11392,42076272,4591024256,4633100528,12395104,4603430752,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +42098816,3264,42102080,4603396976,4645499056,2352544,4605752784,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +42115104,2848,42117952,4605736368,4647854320,3720192,4609459408,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +42120208,3216,42123424,4609454160,4651577584,3872096,4613329472,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +42130048,2976,42133024,4613319472,4655452496,5580512,4618902960,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +42140336,3216,42143552,4618890736,4661034288,5712640,4624606592,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +42151568,2848,42154416,4624594752,4666749168,241404896,4866002496,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +42184416,3072,42187488,4865968688,4908156176,2166912,4868138672,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +42188048,4256,42192304,4868132672,4910324976,2016,4868138944,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +42198960,3168,42202128,4868126944,4910329072,2912,4868133024,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +42207552,3536,42211088,4868122176,4910333264,1536,4868127248,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +42228352,3328,42231680,4868105584,4910337264,48064,4868156976,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +42243984,3664,42247648,4868140784,4910388432,2752704,4870897152,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +42273360,3472,42276832,4870866160,4913142992,1920,4870871552,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +42313360,2880,42316240,4870830880,4913147120,7930720,4878764480,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +42351024,2784,42353808,4878725344,4921079152,5452960,4884181088,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +42365344,3120,42368464,4884166432,4926534896,3469120,4887638672,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +42381936,3200,42385136,4887621216,4930006352,2464,4887626880,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +42389376,3136,42392512,4887617808,4930010320,2016,4887622960,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +42407104,3024,42410128,4887604320,4930014448,46528,4887653872,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +42412112,2880,42414992,4887648608,4930063600,3409568,4891061056,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +42415504,2912,42418416,4891057120,4933475536,1952,4891061984,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +42418944,2592,42421536,4891058128,4933479664,2424224,4893484944,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +42450368,2896,42453264,4893452256,4935905520,1888,4893457040,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +42490640,4240,42494880,4893414768,4935909648,33606624,4927025632,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +42523328,3088,42526416,4926992960,4969519376,139712,4927135760,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +42527904,2960,42530864,4927130816,4969661680,12773024,4939906800,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +42531376,2512,42533888,4939902192,4982436080,2176,4939906880,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +42534288,2912,42537200,4939902976,4982440176,10351520,4950257408,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +42549168,3008,42552176,4950240800,4992792976,1920,4950245728,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +42557632,3136,42560768,4950236144,4992796912,832,4950240112,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +42582272,4560,42586832,4950212224,4992799056,15773088,4965989872,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +42610448,2688,42613136,4965961600,5008574736,5499680,4971463968,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +42624080,3872,42627952,4971448704,5014076656,2496,4971455072,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +42633792,3760,42637552,4971443200,5014080752,1440,4971448400,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +42641376,3296,42644672,4971440176,5014084848,3712,4971447184,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +42648848,11888,42660736,4971430256,5014090992,3584,4971445728,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +42664816,3616,42668432,4971428576,5014097008,1344,4971433536,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +42678416,3136,42681552,4971419840,5014101392,4544,4971427520,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +42686544,2944,42689488,4971417920,5014107408,1088,4971421952,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +42693952,3616,42697568,4971413904,5014111472,1600,4971419120,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +42701728,3088,42704816,4971410720,5014115536,1920,4971415728,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +42709776,3056,42712832,4971406832,5014119664,1248,4971411136,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +42774272,3312,42777584,4971346144,5014123728,26144,4971375600,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +42800128,3248,42803376,4971349120,5014152496,3804640,4975157008,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +42818640,3056,42821696,4975136944,5017958640,22080,4975162080,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +42825472,3040,42828512,4975154672,5017983184,32192,4975189904,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +42843696,2976,42846672,4975171360,5018018032,47840,4975222176,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +42848704,2736,42851440,4975216736,5018068176,3836736,4979056208,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +42851872,2672,42854544,4979051712,5021906256,3744,4979058128,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +42855056,2752,42857808,4979054528,5021912336,2440928,4981498208,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +42891168,3120,42894288,4981460288,5024354576,1856,4981465264,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +42934592,3792,42938384,4981420256,5024358640,24941184,5006365232,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +43012032,3600,43015632,5006285568,5049301200,12450592,5018739760,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +43037520,3328,43040848,5018714496,5061755344,2347648,5021065472,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +43053808,2832,43056640,5021048560,5064105200,3717760,5024769152,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +43058816,2960,43061776,5024762688,5067824464,3725984,5028491632,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +43077152,2800,43079952,5028471744,5071551696,5502592,5033977136,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +43087168,3392,43090560,5033965168,5077055728,5676000,5039644560,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +43098672,3280,43101952,5039632912,5082734864,240707424,5280343616,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +43131232,3136,43134368,5280310128,5323444496,2169536,5282482800,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +43134784,2784,43137568,5282477744,5325615312,1952,5282482480,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +43144112,3472,43147584,5282471856,5325619440,2880,5282478208,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +43152112,2928,43155040,5282468560,5325623600,1536,5282473024,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +43171808,2768,43174576,5282453056,5325627632,47104,5282502928,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +43187280,3328,43190608,5282486144,5325676752,2767680,5285257152,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +43218224,3136,43221360,5285225344,5328446704,1920,5285230400,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +43258944,4080,43263024,5285187776,5328450800,8481312,5293673168,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +43297584,3776,43301360,5293632288,5336933648,5526464,5299162528,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +43311920,3280,43315200,5299146992,5342462192,3485760,5302636032,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +43330368,4160,43334528,5302615408,5345949936,2400,5302621968,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +43338352,2992,43341344,5302612688,5345954032,1984,5302617664,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +43355344,2992,43358336,5302599792,5345958128,46688,5302649472,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +43360288,2992,43363280,5302642944,5346006224,3422112,5306068048,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +43363728,3616,43367344,5306063296,5349430640,1952,5306068864,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +43368000,3360,43371360,5306063216,5349434576,2419200,5308485776,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +43401296,3168,43404464,5308450880,5351855344,1888,5308455936,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +43447696,3296,43450992,5308408416,5351859408,34284960,5342696672,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +43479520,2848,43482368,5342664720,5386147088,139712,5342807280,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +43484000,4368,43488368,5342801024,5386289392,12727072,5355532464,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +43488864,2992,43491856,5355526848,5399018704,1984,5355531824,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +43492304,2896,43495200,5355527664,5399022864,10358368,5365888928,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +43507552,2944,43510496,5365873200,5409383696,1920,5365878064,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +43516336,2816,43519152,5365868608,5409387760,832,5365872256,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +43540528,2896,43543424,5365846448,5409389872,15717632,5381566976,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +43566800,3376,43570176,5381540080,5425110256,5460384,5387003840,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +43581664,3664,43585328,5386986912,5430572240,1952,5386992528,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +43590800,5296,43596096,5386980272,5430576368,1440,5386987008,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +43599648,3680,43603328,5386977136,5430580464,1312,5386982128,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +43607376,3168,43610544,5386974016,5430584560,3392,5386980576,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +43614384,3504,43617888,5386972784,5430590672,1376,5386977664,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +43627120,3712,43630832,5386963968,5430594800,4256,5386971936,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +43635568,3712,43639280,5386961696,5430600976,1088,5386966496,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +43643600,3984,43647584,5386957456,5430605040,1600,5386963040,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +43651664,3360,43655024,5386954080,5430609104,1504,5386958944,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +43659968,3760,43663728,5386949600,5430613328,1248,5386954608,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +43724128,2992,43727120,5386890176,5430617296,25312,5386918480,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +43749936,3456,43753392,5386890560,5430643952,3500224,5390394240,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +43771760,3200,43774960,5390371104,5434146064,2496,5390376800,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +43781168,3520,43784688,5390365504,5434150192,2112,5390371136,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +43799360,3232,43802592,5390351632,5434154224,49952,5390404816,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +43804544,3648,43808192,5390398256,5434206448,4299072,5394700976,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +43808624,2944,43811568,5394696704,5438508272,3648,5394703296,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +43812224,2912,43815136,5394699280,5438514416,2441600,5397143792,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +43848848,3424,43852272,5397106432,5440958704,1888,5397111744,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +43892688,4320,43897008,5397065792,5440962800,24888224,5421958336,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +43978688,3840,43982528,5421869808,5465852336,12351616,5434225264,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +44004944,3200,44008144,5434205792,5478213936,2420480,5436629472,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +44021312,3184,44024496,5436612160,5480636656,3707936,5440323280,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +44026976,2896,44029872,5440316704,5484346576,3735616,5444055216,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +44036448,2768,44039216,5444044960,5488084176,5065568,5449113296,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +44047552,2672,44050224,5449101760,5493151984,5827808,5454932240,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +44057952,3200,44061152,5454920464,5498981616,241799776,5696723440,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +44096032,3168,44099200,5696683696,5740782896,2166496,5698853360,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +44099648,2768,44102416,5698848448,5742950864,1984,5698853200,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +44108928,2864,44111792,5698842912,5742954704,2912,5698848688,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +44116624,2832,44119456,5698839472,5742958928,1536,5698843840,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +44136320,3392,44139712,5698823184,5742962896,46112,5698872688,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +44152480,3328,44155808,5698856240,5743012048,2754048,5701613616,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +44186144,2880,44189024,5701579664,5745768688,1920,5701584464,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +44226496,3056,44229552,5701543200,5745772752,7818272,5709364528,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +44267568,2928,44270496,5709322608,5753593104,5449504,5714775040,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +44281088,3328,44284416,5714760400,5759044816,3466176,5718229904,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +44298464,3984,44302448,5718209856,5762512304,2304,5718216144,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +44306352,3072,44309424,5718206784,5762516208,1952,5718211808,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +44323712,4112,44327824,5718192480,5762520304,47168,5718243760,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +44329552,2576,44332128,5718237328,5762569456,3411264,5721651168,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +44332576,2672,44335248,5721648192,5765983440,1952,5721652816,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +44335776,2576,44338352,5721649216,5765987568,2425440,5724077232,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +44370160,2992,44373152,5724042352,5768415504,1888,5724047232,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +44409296,3216,44412512,5724007056,5768419568,34226240,5758236512,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +44439248,2928,44442176,5758206800,5802648976,140032,5758349760,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +44443744,3120,44446864,5758344288,5802791152,11884448,5770231856,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +44447280,2704,44449984,5770226896,5814676880,1920,5770231520,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +44450496,2864,44453360,5770227488,5814680848,10341376,5780571728,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +44465360,4704,44470064,5780555200,5825025264,1984,5780561888,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +44475600,3792,44479392,5780549968,5825029360,2560,5780556320,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +44501312,2688,44504000,5780529456,5825033456,15733664,5796265808,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +44527440,2752,44530192,5796238208,5840768400,5464736,5801705696,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +44542064,3776,44545840,5801688576,5846234416,1920,5801694272,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +44550864,2816,44553680,5801684800,5846238480,1408,5801689024,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +44557152,2976,44560128,5801682384,5846242512,1312,5801686672,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +44564160,4992,44569152,5801677456,5846246608,3072,5801685520,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +44573088,2736,44575824,5801676928,5846252752,1344,5801681008,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +44584048,3360,44587408,5801669472,5846256880,4864,5801677696,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +44591984,2880,44594864,5801668192,5846263056,1088,5801672160,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +44599088,3312,44602400,5801664752,5846267152,1664,5801669728,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +44606464,3824,44610288,5801660896,5846271184,1504,5801666224,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +44615008,3696,44618704,5801656608,5846275312,1280,5801661584,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +44679024,3312,44682336,5801597040,5846279376,25952,5801626304,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +44706848,2944,44709792,5801598288,5846308080,3495520,5805096752,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +44727696,3312,44731008,5805075024,5849806032,2368,5805080704,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +44734896,4432,44739328,5805070800,5849810128,2112,5805077344,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +44754048,3200,44757248,5805056976,5849814224,48096,5805108272,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +44759184,3328,44762512,5805101920,5849864432,3438592,5808543840,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +44762976,2896,44765872,5808540192,5853306064,3424,5808546512,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +44766416,4032,44770448,5808541920,5853312368,2821952,5811367904,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +44803920,3072,44806992,5811329472,5856136464,1856,5811334400,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +44847104,3072,44850176,5811290352,5856140528,25366272,5836659696,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +44932048,4240,44936288,5836572880,5881509168,11847616,5848424736,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +44958528,3408,44961936,5848398304,5893360240,2496864,5850898576,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +44975088,2880,44977968,5850881472,5895859440,3826176,5854710528,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +44980352,2576,44982928,5854705344,5899688272,3904192,5858612112,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +44989552,3024,44992576,5858602128,5903594704,5218816,5863823968,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +44999552,2656,45002208,5863812880,5908815088,5683744,5869499280,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +45009792,3008,45012800,5869487536,5914500336,241798560,6111289104,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +45041408,3248,45044656,6111256928,6156301584,2171424,6113431600,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +45045120,2832,45047952,6113427552,6158475504,1984,6113432368,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +45054720,3248,45057968,6113421632,6158479600,2912,6113427792,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +45062864,15456,45078320,6113405472,6158483792,1536,6113422464,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +45095600,2976,45098576,6113389216,6158487792,48224,6113440416,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +45111360,2992,45114352,6113424608,6158538960,2755936,6116183536,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +45139568,2976,45142544,6116155104,6161297648,1952,6116160032,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +45178976,2816,45181792,6116119920,6161301712,7908928,6124031664,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +45217472,3008,45220480,6123992720,6169213200,5461504,6129457232,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +45230960,3536,45234496,6129442704,6174677200,3478752,6132924992,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +45249680,3312,45252992,6132904816,6178157808,2304,6132910432,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +45256880,4208,45261088,6132900816,6178161904,2016,6132907040,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +45275440,2896,45278336,6132887664,6178166000,45856,6132936416,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +45280112,3344,45283456,6132930672,6178214128,3426912,6136360928,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +45283920,2768,45286688,6136356784,6181643472,1984,6136361536,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +45287168,2848,45290016,6136357584,6181647600,2425184,6138785616,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +45319184,3152,45322336,6138753168,6184075504,1888,6138758208,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +45359376,3840,45363216,6138716384,6184079600,34891392,6173611616,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +45390320,3296,45393616,6173578816,6218972432,141280,6173723392,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +45395152,4240,45399392,6173717392,6219116784,11920992,6185642624,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +45399808,3120,45402928,6185637312,6231040240,1952,6185642384,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +45403504,3584,45407088,6185637248,6231044336,10766528,6196407360,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +45419680,3792,45423472,6196390272,6241813744,2144,6196396208,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +45428880,2832,45431712,6196386128,6241817840,832,6196389792,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +45453056,3504,45456560,6196363392,6241819952,16326752,6212693648,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +45479728,2768,45482496,6212666096,6258148592,5467264,6218136128,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +45493872,3264,45497136,6218120640,6263617776,1984,6218125888,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +45502512,2768,45505280,6218116624,6263621904,1408,6218120800,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +45508992,2928,45511920,6218114048,6263625968,1312,6218118288,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +45515936,3072,45519008,6218111056,6263630064,3264,6218117392,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +45522928,2848,45525776,6218110432,6263636208,1440,6218114720,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +45533888,4496,45538384,6218101920,6263640304,4704,6218111120,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +45543088,2736,45545824,6218100656,6263646480,1088,6218104480,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +45550000,2688,45552688,6218097856,6263650544,1600,6218102144,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +45556608,2880,45559488,6218095152,6263654640,1472,6218099504,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +45564208,13216,45577424,6218081312,6263658736,1280,6218095808,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +45638752,3008,45641760,6218021072,6263662832,26688,6218050768,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +45666032,2832,45668864,6218022640,6263691504,3502624,6221528096,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +45690176,3312,45693488,6221502144,6267195632,2400,6221507856,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +45697472,3152,45700624,6221499072,6267199696,1984,6221504208,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +45715520,3536,45719056,6221484512,6267203568,47584,6221535632,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +45721232,3152,45724384,6221529616,6267254000,3522112,6225054880,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +45724816,2768,45727584,6225051024,6270778608,3968,6225057760,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +45728112,3392,45731504,6225053248,6270784752,2439168,6227495808,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +45767856,3760,45771616,6227455376,6273226992,1888,6227461024,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +45813360,3168,45816528,6227414560,6273231088,25829856,6253247584,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +45893152,3840,45896992,6253165520,6299062512,11811904,6264981264,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +45918736,3424,45922160,6264955168,6310877328,2378688,6267337280,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +45934992,2944,45937936,6267319616,6313257552,3908832,6271231392,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +45940288,2832,45943120,6271224832,6317167952,3972352,6275200016,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +45949744,2768,45952512,6275190960,6321143472,5239040,6280432768,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +45959552,2736,45962288,6280422752,6326385040,5665728,6286091216,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +45970176,3600,45973776,6286078944,6332052720,240203808,6526286352,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +46001904,3280,46005184,6526254416,6572259600,2177760,6528435456,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +46005776,2704,46008480,6528431184,6574439664,1952,6528435840,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +46015264,3520,46018784,6528424944,6574443728,2976,6528431440,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +46023680,3216,46026896,6528421088,6574447984,1504,6528425808,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +46043904,3088,46046992,6528404928,6574451920,48864,6528456880,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +46059200,3680,46062880,6528439312,6574502192,2758368,6531201360,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +46096784,3168,46099952,6531162880,6577262832,1920,6531167968,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +46137232,2784,46140016,6531126912,6577266928,7814496,6538944192,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +46175328,4000,46179328,6538904848,6585084176,5459264,6544368112,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +46190976,3280,46194256,6544351904,6590546160,3478464,6547833648,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +46207680,3472,46211152,6547814752,6594025904,2400,6547820624,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +46215216,2944,46218160,6547811648,6594029808,2112,6547816704,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +46232528,2624,46235152,6547798752,6594033904,47328,6547848704,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +46236912,2944,46239856,6547843168,6594083024,3426688,6551272800,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +46240352,2880,46243232,6551269200,6597512432,1952,6551274032,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +46243920,2640,46246560,6551269968,6597516528,2426080,6553698688,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +46276864,2928,46279792,6553664672,6599944464,1856,6553669456,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +46316976,2912,46319888,6553628640,6599948528,33846624,6587478176,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +46349280,3312,46352592,6587445280,6633797872,172064,6587620656,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +46354064,3120,46357184,6587615760,6633972944,12456320,6600075200,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +46357584,3136,46360720,6600071232,6646431952,2208,6600076576,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +46361200,2928,46364128,6600071952,6646436080,10598688,6610673568,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +46376480,2832,46379312,6610657344,6657036656,1952,6610662128,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +46384752,3520,46388272,6610652384,6657040656,864,6610656768,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +46410256,2736,46412992,6610629840,6657042832,15399488,6626032064,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +46442944,3472,46446416,6625998240,6672444656,5615488,6631617200,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +46459536,4016,46463552,6631597904,6678061456,2016,6631603936,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +46469792,4016,46473808,6631591648,6678065456,1664,6631597328,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +46477456,3120,46480576,6631589168,6678069744,1312,6631593600,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +46484752,3232,46487984,6631585600,6678073584,3648,6631592480,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +46492128,3088,46495216,6631584480,6678079696,1344,6631588912,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +46503712,3120,46506832,6631576992,6678083824,11104,6631591216,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +46511520,2768,46514288,6631581920,6678096208,1856,6631586544,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +46518816,3056,46521872,6631578336,6678100208,2080,6631583472,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +46525936,2704,46528640,6631575664,6678104304,4032,6631582400,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +46533408,3984,46537392,6631573056,6678110448,1504,6631578544,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +46601728,3792,46605520,6631509024,6678114544,62208,6631575024,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +46628592,3440,46632032,6631546960,6678178992,3617312,6635167712,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +46648448,3280,46651728,6635147200,6681798928,2560,6635153040,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +46655872,2816,46658688,6635144336,6681803024,2080,6635149232,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +46673568,3632,46677200,6635129856,6681807056,48160,6635181648,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +46679248,2912,46682160,6635175104,6681857264,3838528,6639016544,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +46683024,3328,46686352,6639011936,6685698288,4000,6639019264,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +46686960,3136,46690096,6639014336,6685704432,2440192,6641457664,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +46723408,4752,46728160,6641418512,6688146672,1920,6641425184,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +46768544,3616,46772160,6641378608,6688150768,26015712,6667397936,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +46856448,3904,46860352,6667309232,6714169584,12030144,6679343280,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +46882336,3312,46885648,6679318336,6726203984,2354880,6681676528,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +46898928,3200,46902128,6681658752,6728560880,3711456,6685373408,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +46904528,4048,46908576,6685366352,6732274928,3963872,6689334272,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +46915360,3296,46918656,6689323216,6736241872,5437760,6694764272,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +46925888,3024,46928912,6694752448,6741681360,5725024,6700480496,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +46936784,3664,46940448,6700468176,6747408624,239633120,6940104960,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +46968976,3360,46972336,6940071776,6987044112,2163136,6942238272,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +46972944,2816,46975760,6942233024,6989208784,1952,6942237792,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +46981776,4736,46986512,6942226432,6989212944,2688,6942233856,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +46991440,2864,46994304,6942222736,6989217040,1504,6942227104,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +47011248,3120,47014368,6942206736,6989221104,46848,6942256704,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +47026912,3472,47030384,6942239968,6989270352,2756896,6945000336,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +47057792,2976,47060768,6944968112,6992028880,1952,6944973040,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +47106496,3168,47109664,6944923376,6992033040,7841664,6952768208,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +47147552,2928,47150480,6952725568,6999876048,5460608,6958189104,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +47160704,3664,47164368,6958174496,7005338864,3489408,6961667568,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +47178480,3328,47181808,6961648896,7008830704,2496,6961654720,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +47186000,3536,47189536,6961645232,7008834768,1984,6961650752,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +47211248,3024,47214272,6961624592,7008838864,46144,6961673760,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +47216032,3008,47219040,6961667952,7008886992,3410336,6965081296,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +47219488,2800,47222288,6965077696,7012299984,1952,6965082448,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +47222912,2784,47225696,6965078416,7012304112,2667616,6967748816,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +47256576,3376,47259952,6967714496,7014974448,3712,6967721584,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +47298000,4288,47302288,6967678560,7014980848,34079488,7001762336,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +47334512,3328,47337840,7001724864,7049062704,139456,7001867648,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +47339248,3456,47342704,7001861248,7049203952,12768992,7014633696,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +47343248,3936,47347184,7014628064,7061975248,1984,7014633984,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +47347664,4256,47351920,7014627456,7061979376,10367616,7024999328,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +47365296,3888,47369184,7024979312,7072348496,1920,7024985120,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +47374768,4288,47379056,7024973440,7072352496,864,7024978592,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +47402384,2704,47405088,7024949552,7072354640,15705376,7040657632,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +47428528,2752,47431280,7040630400,7088061680,5511744,7046144896,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +47443024,3280,47446304,7046128656,7093574960,2272,7046134208,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +47451792,3952,47455744,7046123248,7093578992,1408,7046128608,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +47459664,3392,47463056,7046120032,7093583088,3680,7046127104,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +47467568,4512,47472080,7046117120,7093589200,3616,7046125248,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +47476176,2976,47479152,7046116096,7093595248,1408,7046120480,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +47487248,3360,47490608,7046108864,7093599472,4448,7046116672,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +47495664,2592,47498256,7046107360,7093605616,1088,7046111040,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +47502976,2848,47505824,7046103920,7093609744,1664,7046108432,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +47509968,3104,47513072,7046100736,7093613808,1472,7046105312,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +47517648,3184,47520832,7046097104,7093617936,1280,7046101568,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +47582576,4016,47586592,7046035408,7093622000,26848,7046066272,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +47610544,3280,47613824,7046036816,7093650640,3741248,7049781344,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +47632480,3600,47636080,7049758336,7097394416,13184,7049775120,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +47640096,2944,47643040,7049766096,7097409136,13472,7049782512,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +47657824,3104,47660928,7049764208,7097425136,97696,7049865008,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +47662976,3776,47666752,7049858736,7097525488,3970560,7053833072,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +47667184,3152,47670336,7053828240,7101498576,3552,7053834944,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +47670816,2976,47673792,7053830960,7101504752,2437920,7056271856,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +47707648,3632,47711280,7056233664,7103944944,1888,7056239184,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +47752112,5072,47757184,7056191856,7103949040,24860704,7081057632,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +47833824,3712,47837536,7080974192,7128811728,12400832,7093378736,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +47859888,3712,47863600,7093351776,7141215376,2338976,7095694464,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +47877008,2992,47880000,7095677328,7143557328,3707360,7099387680,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +47882272,3328,47885600,7099381680,7147267280,3739648,7103124656,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +47892112,3472,47895584,7103113392,7151008976,5512160,7108629024,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +47902800,4768,47907568,7108614848,7156522416,5684192,7114303808,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +47916352,4000,47920352,7114289200,7162209552,242227488,7356520688,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +47955696,3872,47959568,7356479328,7404438896,2180384,7358663584,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +47960080,2544,47962624,7358658256,7406620880,1952,7358662752,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +47969216,3056,47972272,7358652704,7406624976,2560,7358658320,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +47976928,2752,47979680,7358649392,7406629072,1536,7358653680,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +47996096,3248,47999344,7358633856,7406633200,45024,7358682128,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +48011680,3824,48015504,7358664800,7406680304,2751680,7361420304,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +48040928,3072,48044000,7361389840,7409433840,1856,7361394768,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +48088768,3936,48092704,7361345232,7409437936,7852832,7369202000,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +48127616,2480,48130096,7369161984,7417292080,5460768,7374625232,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +48140272,3568,48143840,7374611216,7422755056,3467584,7378082368,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +48157952,3584,48161536,7378063856,7426225392,2368,7378069808,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +48165648,4000,48169648,7378059840,7426229488,2112,7378065952,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +48183984,3280,48187264,7378046352,7426233616,45408,7378095040,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +48189056,2848,48191904,7378089808,7426281712,3408320,7381500976,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +48192384,2816,48195200,7381496400,7429691600,1952,7381501168,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +48195728,2512,48198240,7381497488,7429695728,2422752,7383922752,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +48227520,3168,48230688,7383890896,7432121584,1856,7383895920,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +48267808,2784,48270592,7383855056,7432125648,34878368,7418736208,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +48298304,2928,48301232,7418703840,7467005072,142560,7418849328,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +48302704,3760,48306464,7418844112,7467150576,11880896,7430728768,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +48307056,2720,48309776,7430723296,7479033072,1952,7430727968,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +48310240,3904,48314144,7430723024,7479037168,10359680,7441086608,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +48326832,2960,48329792,7441069296,7489399088,6624,7441078880,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +48335312,3488,48338800,7441068416,7489407216,1568,7441073472,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +48361296,2944,48364240,7441047072,7489411312,15720800,7456770816,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +48387760,2720,48390480,7456744384,7505134864,5521792,7462268896,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +48405280,3264,48408544,7462250768,7510659312,3904,7462257936,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +48414048,3120,48417168,7462248256,7510665424,1440,7462252816,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +48420864,2912,48423776,7462245776,7510669552,3008,7462251696,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +48427840,3168,48431008,7462244688,7510675696,3520,7462251376,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +48435056,2992,48438048,7462243728,7510681776,2048,7462248768,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +48447040,5072,48452112,7462233824,7510685936,6688,7462245584,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +48456896,3600,48460496,7462233632,7510694128,2848,7462240080,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +48464656,4368,48469024,7462229264,7510698288,1600,7462235232,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +48473168,3184,48476352,7462225968,7510702320,2304,7462231456,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +48481088,3184,48484272,7462222144,7510706416,2240,7462227568,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +48545280,2880,48548160,7462162352,7510710512,29504,7462194736,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +48573248,2992,48576240,7462165056,7510741296,3520864,7465688912,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +48592848,2864,48595712,7465669392,7514265104,5600,7465677856,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +48599600,2864,48602464,7465669520,7514271984,1952,7465674336,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +48617440,3008,48620448,7465655504,7514275952,52032,7465710544,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +48622416,3472,48625888,7465704528,7514330416,4260224,7469968224,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +48626368,2960,48629328,7469962880,7518592208,1984,7469967824,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +48629824,3136,48632960,7469963376,7518596336,2447712,7472414224,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +48670864,3104,48673968,7472372832,7521046800,1856,7472377792,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +48723920,3072,48726992,7472323616,7521050608,25226208,7497552896,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +48804096,3680,48807776,7497470320,7546278096,12453792,7509927792,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +48831328,3248,48834576,7509900896,7558735472,2351712,7512255856,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +48848192,3264,48851456,7512238832,7561090288,3714304,7515956400,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +48853744,2640,48856384,7515950992,7564807376,3786336,7519739968,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +48863008,2768,48865776,7519730400,7568596176,5154144,7524887312,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +48873008,2720,48875728,7524877344,7573753072,5813440,7530693504,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +48883664,3088,48886752,7530681616,7579568368,240922976,7771607680,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +48918176,3008,48921184,7771571888,7820493072,2156896,7773731792,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +48921664,2784,48924448,7773727216,7822651664,1952,7773731952,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +48931216,3200,48934416,7773721312,7822655728,3360,7773727872,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +48939184,3024,48942208,7773719632,7822661840,1568,7773724224,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +48958992,2896,48961888,7773703984,7822665872,51264,7773758144,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +48974352,3440,48977792,7773741392,7822719184,2771712,7776516544,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +49003200,2976,49006176,7776485968,7825492144,1952,7776490896,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +49042816,4128,49046944,7776449360,7825496304,7849312,7784302800,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +49090640,3392,49094032,7784253312,7833347344,5451136,7789707840,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +49104256,4160,49108416,7789691824,7838800240,3479136,7793175120,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +49122672,3456,49126128,7793154528,7842280656,2368,7793160352,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +49130368,3104,49133472,7793151280,7842284752,2144,7793156528,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +49148048,2976,49151024,7793137856,7842288880,47424,7793188256,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +49152784,3008,49155792,7793183264,7842339056,3417600,7796603872,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +49156272,3008,49159280,7796598880,7845758160,2016,7796603904,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +49159808,3152,49162960,7796599328,7845762288,2423840,7799026320,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +49193792,2928,49196720,7798992448,7848189168,1888,7798997264,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +49235024,3184,49238208,7798954800,7848193008,34270688,7833228672,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +49265936,2944,49268880,7833196672,7882465552,140768,7833340384,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +49270336,3232,49273568,7833334288,7882607856,11865984,7845203504,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +49274224,3792,49278016,7845197104,7894475120,2464,7845203360,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +49278608,2928,49281536,7845197552,7894479088,10348864,7855549344,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +49293728,4224,49297952,7855532752,7904830704,1952,7855538928,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +49303280,2912,49306192,7855528608,7904834800,2656,7855534176,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +49329184,3088,49332272,7855506624,7904838896,15733856,7871243568,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +49355536,2560,49358096,7871217632,7920575728,5463552,7876683744,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +49370032,3536,49373568,7876668272,7926041840,1920,7876673728,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +49378896,3088,49381984,7876663920,7926045904,1440,7876668448,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +49385776,4128,49389904,7876660096,7926050000,1312,7876665536,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +49393696,2992,49396688,7876657408,7926054096,3520,7876663920,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +49400864,2912,49403776,7876656496,7926060272,1344,7876660752,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +49412160,4224,49416384,7876647984,7926064368,4992,7876657200,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +49421152,2960,49424112,7876646528,7926070640,1120,7876650608,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +49428624,2992,49431616,7876643024,7926074640,1632,7876647648,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +49435632,2832,49438464,7876640240,7926078704,1504,7876644576,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +49443168,3296,49446464,7876636336,7926082800,1216,7876640848,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +49516688,3232,49519920,7876566976,7926086896,25632,7876595840,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +49541744,2992,49544736,7876570832,7926115568,3505824,7880079648,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +49562112,2960,49565072,7880058720,7929623792,2432,7880064112,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +49569008,2848,49571856,7880056032,7929627888,1920,7880060800,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +49587456,3328,49590784,7880041200,7929631984,47648,7880092176,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +49592704,11806400,61399104,7868282032,7929681136,3448928,7883537360,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +61399680,11859920,73259600,7859873440,7933133040,4032,7871737392,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +73260240,3024,73263264,7859875984,7933139248,2809952,7862688960,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +73307664,2350480,75658144,7860293968,7935952112,1856,7862646304,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +75703072,3960864,79663936,7856292336,7935956272,25293632,7885546832,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +79746944,3917648,83664592,7877587488,7961252080,11798880,7893304016,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +88963632,5733888,94697520,7878357152,7973054672,2382432,7886473472,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +94711456,240306272,335017728,7640421872,7975439600,3919904,7884648048,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +335024416,2168400,337192816,7642169728,7979362544,3922944,7648261072,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +337203344,3712,337207056,7646080448,7983287504,5215584,7651299744,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +337215072,2800,337217872,7651287936,7988505808,5690784,7656981520,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +337230096,3168,337233264,7656967392,7994200656,241148000,7898118560,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +337294976,6464,337301440,7898049872,8235351312,2184512,7900240848,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +337301936,2723280,340025216,7897513296,8237538512,1984,7900238560,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +340035552,4928,340040480,7897502160,8237542640,3040,7897510128,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +340045936,7686896,347732832,7889815920,8237548752,1568,7897504384,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +347761408,5435120,353196528,7884356256,8237552784,52064,7889843440,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +353214192,3701616,356915808,7880690320,8237606128,2756640,7887148576,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +356948160,4160,356952320,7883413488,8240365808,1888,7883419536,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +356996944,3296,357000240,7883369664,8240369904,7952640,7891325600,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +357046496,5888,357052384,7891271984,8248324368,5449216,7896727088,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +357064880,4020800,361085680,7892689408,8253775088,3472800,7900183008,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +361106736,3728,361110464,7896139056,8257249520,2400,7896145184,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +361114448,2413456,363527904,7893725712,8257253616,1984,7896141152,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +363545184,5984,363551168,7893706544,8257257712,45696,7893758224,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +363553104,34544576,398097680,7859208160,8257305840,3431232,7897183968,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +398098240,141808,398240048,7862499232,8260739280,1952,7862642992,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +398240656,11924096,410164752,7850578720,8260743472,2425536,7864928352,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +410204272,4128,410208400,7852962912,8263171312,1888,7852968928,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +410248736,10679792,420928528,7842246880,8263175408,34385728,7887312400,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +420958736,4240,420962976,7876599472,8297562448,140160,7876743872,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +420964432,4400,420968832,7876736880,8297705712,11916032,7888657312,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +420969408,16172368,437141776,7872482272,8309624048,2432,7888657072,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +437142592,5455488,442598080,7867030064,8309628144,10800032,7883285584,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +442612544,3056,442615600,7877814720,8320430320,1920,7877819696,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +442622224,2896,442625120,7877809296,8320434416,832,7877813024,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +442652112,3360,442655472,7877781152,8320436624,16112576,7893897088,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +442684560,5488,442690048,7893861104,8336551152,5465952,7899332544,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +442708112,3952,442712064,7899307248,8342019312,1952,7899313152,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +442718272,3408,442721680,7899301760,8342023440,1440,7899306608,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +442728144,5200,442733344,7899294192,8342027536,1312,7899300704,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +442738480,6032,442744512,7899287088,8342031600,3680,7899296800,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +442749424,4256,442753680,7899284096,8342037776,1312,7899289664,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +442768256,7120,442775376,7899266432,8342041808,4896,7899278448,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +442784720,4832,442789552,7899258432,8342047984,1088,7899264352,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +442793920,3362736,446156656,7895895424,8342052080,1632,7899259792,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +446161632,3696,446165328,7895890848,8342056176,1472,7895896016,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +446172096,3840,446175936,7895884336,8342060272,1248,7895889424,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +446247728,3776,446251504,7895812864,8342064368,26112,7895842752,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +446277584,3390880,449668464,7892424576,8342093040,3493632,7899309088,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +449690320,3728,449694048,7895893904,8345587952,2560,7895900192,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +449698016,2414112,452112128,7893479888,8345592016,1984,7895895984,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +452128752,4144,452132896,7893463248,8345596144,47840,7893515232,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +452135120,25754352,477889472,7867756848,8345646320,3444576,7896955776,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +477889952,11822032,489711984,7859380224,8349092208,4096,7871206352,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +489712736,5920,489718656,7859379568,8349098224,2448192,7861833680,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +489767792,2294704,492062496,7859486160,8351548656,1888,7861782752,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +492114336,3760928,495875264,7855677520,8351552784,25738272,7885176720,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +495967680,3856864,499824544,7877471088,8377295632,11891808,7893219760,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +505233904,5665120,510899024,7878291168,8389190192,2346944,7886303232,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +510913488,240436896,751350384,7640188544,8391538928,3761664,7884387104,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +751355552,2164752,753520304,7641782848,8395303152,4026528,7647974128,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +753528464,3264,753531728,7645799808,8399331536,5417472,7651220544,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +753539184,3008,753542192,7651209408,8404751600,5663296,7656875712,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +753552496,3328,753555824,7656860544,8410416368,241593728,7898457600,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +753597488,7360,753604848,7898406944,8652011792,2164960,7900579264,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +753605360,2726944,756332304,7897846240,8654178544,1920,7900575104,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +756341584,6816,756348400,7897834240,8654182640,2976,7897844032,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +756353632,7728624,764082256,7890104640,8654186896,1536,7897834800,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +764105344,5441520,769546864,7884643968,8654190832,48416,7890133904,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +769561696,3448912,773010608,7881231424,8654242032,2752160,7887432496,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +773041040,4144,773045184,7883950384,8656995568,1888,7883956416,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +773088336,5424,773093760,7883905904,8656999664,7942144,7891853472,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +773132160,3328,773135488,7891808432,8664943920,5458208,7897269968,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +773146880,3889008,777035888,7893368960,8670404848,3475616,7900733584,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +777056048,3504,777059552,7896823824,8673883376,4320,7896831648,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +777063680,2707872,779771552,7894118128,8673889680,19776,7896845776,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +779787840,3856,779791696,7894120320,8673912016,52288,7894176464,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +779793584,33119872,812913456,7861053856,8673967312,4322208,7898495936,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +812914032,187792,813101824,7865189872,8678291696,1952,7865379616,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +813102480,12696720,825799200,7852496560,8678295760,2444640,7867637920,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +825834960,5632,825840592,7854902560,8680743152,1920,7854910112,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +825881168,10399248,836280416,7844466832,8680747248,34577664,7889443744,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +836312144,3328,836315472,7879011232,8715326704,141088,7879155648,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +836317184,3008,836320192,7879148880,8715469072,12382976,7891534864,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +836320688,15581440,851902128,7875951264,8727853392,1984,7891534688,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +851902672,5573968,857476640,7870380720,8727857360,10443584,7886398272,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +857490592,3136,857493728,7880809488,8738303216,1952,7880814576,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +857499872,2736,857502608,7880804704,8738307312,2592,7880810032,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +857525312,2944,857528256,7880783152,8738311408,15308832,7896094928,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +857554704,2960,857557664,7896064592,8753622256,5479104,7901546656,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +857569936,3104,857573040,7901529664,8759102704,1952,7901534720,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +857578896,2768,857581664,7901525136,8759106800,1408,7901529312,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +857585392,2784,857588176,7901522720,8759110896,3648,7901529152,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +857592800,3616,857596416,7901520624,8759117040,3072,7901527312,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +857600800,2976,857603776,7901519408,8759123184,1376,7901523760,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +857616288,10272,857626560,7901500624,8759127184,4640,7901515536,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +857631712,6768,857638480,7901494976,8759133456,1152,7901502896,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +857644512,3638560,861283072,7897854192,8759137264,1600,7901494352,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +861287728,3328,861291056,7897850528,8759141584,1472,7897855328,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +861296528,3264,861299792,7897845920,8759145712,1248,7897850432,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +861370064,3376,861373440,7897776336,8759149776,25216,7897804928,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +861398816,3758960,865157776,7894018656,8759176432,3694624,7901472240,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +865177456,3296,865180752,7897693344,8762874096,2400,7897699040,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +865184752,2416176,867600928,7895277328,8762878256,2176,7897695680,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +867617840,3488,867621328,7895260960,8762882288,50528,7895314976,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +867623760,23921488,891545248,7871389232,8762934480,3638240,7898948960,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +891545968,12482496,904028464,7862546336,8766574800,1952,7875030784,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +904029136,4048,904033184,7862545744,8766578928,2442912,7864992704,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +904078192,2298928,906377120,7862647248,8769024368,1920,7864948096,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +906426704,3665792,910092496,7858935872,8769028368,25552160,7888153824,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +910192640,3624192,913816832,7880765456,8794582288,12077248,7896466896,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +919352704,5796848,925149552,7881512544,8806662096,2352160,7889661552,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +925163920,239341408,1164505328,7644510240,8809015568,3721376,7887573024,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +1164510912,2162880,1166673792,7646064976,8812738768,4098432,7652326288,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +1166682640,3280,1166685920,7650153968,8816839888,5346304,7655503552,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +1166693600,2816,1166696416,7655492848,8822189264,5725184,7661220848,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +1166708128,3216,1166711344,7661205184,8827916528,241606016,7902814416,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +1166759440,8128,1166767568,7902757696,9069525264,2174688,7904940512,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +1166768208,2728592,1169496800,7902205520,9071702320,1952,7904936064,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +1169508624,4384,1169513008,7902193376,9071706384,2720,7902200480,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +1169518000,7785488,1177303488,7894406928,9071710416,1536,7902193952,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +1177330704,5426464,1182757168,7888957344,9071714512,47232,7894431040,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +1182776848,3458304,1186235152,7885528512,9071763664,2758720,7891745536,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +1186266928,4224,1186271152,7888253248,9074524400,1952,7888259424,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +1186319216,3120,1186322336,7888206160,9074528496,7921504,7896130784,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +1186364320,3824,1186368144,7896083200,9082451344,5461920,7901548944,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +1186381648,3372992,1189754640,7898160608,9087915248,3486688,7905020288,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +1189775840,4432,1189780272,7901624768,9091405040,2400,7901631600,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +1189784432,2427904,1192212336,7899196800,9091409136,2080,7901626784,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +1192229088,4480,1192233568,7899179632,9091413200,46048,7899230160,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +1192235504,34629104,1226864608,7864597744,9091462352,3874528,7903101376,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +1226865184,142416,1227007600,7868331648,9095339248,1952,7868476016,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +1227008240,12762544,1239770784,7855572560,9095343344,2624832,7870959936,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +1239810080,4960,1239815040,7858155888,9097970928,1856,7858162704,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +1239856560,10376672,1250233232,7847741792,9097975024,33697120,7891815584,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +1250265872,4592,1250270464,7881404432,9131674896,139840,7881548864,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +1250272576,4880,1250277456,7881539744,9131817200,11919712,7893464336,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +1250278192,15756848,1266035040,7877704560,9143739600,1952,7893463360,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +1266036336,5823472,1271859808,7871883952,9143743760,10381664,7888089088,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +1271875280,3280,1271878560,7882250384,9154128944,1952,7882255616,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +1271885296,2800,1271888096,7882245008,9154133104,1824,7882249632,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +1271915792,4544,1271920336,7882215968,9154136304,15711680,7897932192,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +1271948400,6048,1271954448,7897896160,9169850608,5507136,7903409344,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +1271974016,3888,1271977904,7903381824,9175359728,1920,7903387632,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +1271984176,6096,1271990272,7903373552,9175363824,1568,7903381216,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +1271994864,4976,1271999840,7903368080,9175367920,3168,7903376224,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +1272004384,6688,1272011072,7903363024,9175374096,3392,7903373104,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +1272016304,7840,1272024144,7903356000,9175380144,1344,7903365184,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +1272038832,8480,1272047312,7903337024,9175384336,4928,7903350432,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +1272054176,5904,1272060080,7903330656,9175390736,1120,7903337680,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +1272064336,3417328,1275481664,7899912784,9175394448,1632,7903331744,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +1275486768,4320,1275491088,7899907552,9175398640,1696,7899913568,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +1275498016,3408,1275501424,7899901344,9175402768,7616,7899912368,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +1275572976,3696,1275576672,7899836304,9175412976,26336,7899866336,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +1275602496,4198528,1279801024,7895640496,9175441520,3844544,7903683568,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +1279823024,3424,1279826448,7899461344,9179287792,2336,7899467104,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +1279830320,2423072,1282253392,7897038496,9179291888,2048,7899463616,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +1282270336,4432,1282274768,7897021216,9179295984,48768,7897074416,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +1282276992,24866304,1307143296,7872203888,9179347184,3828800,7900898992,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +1307143776,12477136,1319620912,7863557056,9183177968,4032,7876038224,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +1319621712,3808,1319625520,7863558592,9183184112,2423808,7865986208,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +1319668752,2294880,1321963632,7863647392,9185611024,1888,7865944160,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +1322013136,3678496,1325691632,7859923424,9185615056,24620416,7888222336,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +1325793664,3633296,1329426960,7880811232,9210238192,12466528,7896911056,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +1334578096,5874496,1340452592,7882255168,9222707760,2350880,7890480544,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +1340468096,238892736,1579360832,7645700784,9225061616,3725056,7888318576,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +1579364448,2156080,1581520528,7647267424,9228787952,3735968,7653159472,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +1581528416,3232,1581531648,7650993872,9232525520,5525984,7656523088,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +1581539200,2416,1581541616,7656511488,9238053104,5802432,7662316336,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +1581551008,2944,1581553952,7662304176,9243858128,242206816,7904513936,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +1581587376,5152,1581592528,7904474432,9486066960,2168960,7906648544,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +1581592960,2744896,1584337856,7903899984,9488237840,2016,7906646896,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +1584346400,4112,1584350512,7903891392,9488241904,2944,7903898448,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +1584355856,7793680,1592149536,7896096592,9488246128,1536,7903891808,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +1592172256,5592352,1597764608,7890485488,9488250096,45504,7896123344,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +1597779344,3648176,1601427520,7886869648,9488297168,2763104,7893280928,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +1601457584,5504,1601463088,7889599936,9491063024,1920,7889607360,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +1601502528,3136,1601505664,7889561456,9491067120,7944544,7897509136,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +1601544480,2784,1601547264,7897467152,9499014416,5450496,7902920432,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +1601558800,3786304,1605345104,7899121088,9504466192,3475360,7906382752,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +1605362560,3504,1605366064,7902576736,9507942800,2368,7902582608,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +1605370032,2420304,1607790336,7900156400,9507946736,2048,7902578752,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +1607806608,3264,1607809872,7900140960,9507950832,47552,7900191776,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +1607811776,34027600,1641839376,7866160608,9507999984,3421152,7903609360,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +1641840048,140432,1641980480,7869443792,9511424272,1952,7869586176,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +1641981120,12368192,1654349312,7857079024,9511428336,2483008,7871930224,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +1654384384,4352,1654388736,7859523920,9513912656,2368,7859530640,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +1654430064,10593680,1665023744,7848893104,9513916848,34376960,7893863744,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +1665054544,4224,1665058768,7883237696,9548296464,139616,7883381536,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +1665060400,4352,1665064752,7883374016,9548438768,12796832,7896175200,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +1665065200,15672688,1680737888,7880499824,9561237712,2240,7896174752,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +1680738400,5455632,1686194032,7875047808,9561241840,10357696,7890861136,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +1686208144,3024,1686211168,7885390480,9571601648,1920,7885395424,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +1686217616,3264,1686220880,7885384864,9571605744,832,7885388960,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +1686244048,3008,1686247056,7885360800,9571607856,15753632,7901117440,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +1686273504,4832,1686278336,7901084720,9587363056,5458112,7906547664,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +1686296208,4416,1686300624,7906522400,9592823024,1984,7906528800,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +1686306352,4080,1686310432,7906516784,9592827216,1408,7906522272,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +1686314784,4704,1686319488,7906511728,9592831216,1312,7906517744,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +1686323696,4992,1686328688,7906506624,9592835312,3296,7906514912,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +1686333600,4336,1686337936,7906503520,9592841456,1344,7906509200,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +1686348208,6320,1686354528,7906490928,9592845456,4448,7906501696,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +1686360256,5248,1686365504,7906486160,9592851664,1088,7906492496,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +1686370368,3391328,1689761696,7903094000,9592855696,1632,7906486960,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +1689766336,3504,1689769840,7903090048,9592859888,1472,7903095024,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +1689775344,5264,1689780608,7903083376,9592863984,1280,7903089920,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +1689847168,3584,1689850752,7903017328,9592868080,25184,7903046096,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +1689876800,3862112,1693738912,7899155792,9592894704,3628416,7906646320,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +1693766128,3248,1693769376,7902764528,9596533904,17728,7902785504,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +1693773536,2819920,1696593456,7899961024,9596554480,11680,7902792624,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +1696610752,4016,1696614768,7899954080,9596568848,69088,7900027184,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +1696616848,25285088,1721901936,7874737536,9596639472,4118272,7904140896,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +1721902592,11903024,1733805616,7866954400,9600760016,3840,7878861264,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +1733806272,2928,1733809200,7866957024,9600766224,2419840,7869379792,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +1733851440,2508640,1736360080,7866827872,9603187952,1888,7869338400,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +1736407728,3696624,1740104352,7863087696,9603192048,24930496,7891714816,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +1740194480,3860432,1744054912,7884069488,9628124400,12466368,7900396288,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +1749176464,5699216,1754875680,7885718000,9640593680,2350944,7893768160,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +1754890528,240183200,1995073728,7647873104,9642946832,3729024,7891785328,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +1995079392,2160912,1997240304,7649437920,9646678224,3748576,7655347408,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +1997248592,3360,1997251952,7653176160,9650428112,5354496,7658534016,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +1997259936,3008,1997262944,7658521776,9655784720,5783808,7664308592,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +1997272656,3040,1997275696,7664295616,9661571312,241193152,7905491808,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +1997309504,8512,1997318016,7905448368,9902766384,2168640,7907625520,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +1997318640,2735392,2000054032,7904883168,9904937200,1952,7907620512,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +2000062368,5552,2000067920,7904873376,9904941296,2944,7904881872,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +2000073360,7796944,2007870304,7897075216,9904945520,1568,7904873728,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +2007892000,5452368,2013344368,7891605088,9904949456,47232,7897104688,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2013360384,3693200,2017053584,7887945056,9904998640,2753088,7894391344,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +2017085744,3984,2017089728,7890664496,9907754224,1888,7890670368,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +2017130592,3728,2017134320,7890624000,9907758320,7844192,7898471920,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +2017175040,4112,2017179152,7898426112,9915605264,5462816,7903893040,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +2017191040,3990144,2021181184,7899889168,9921070352,3466848,7907346160,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +2021201840,3728,2021205568,7903334064,9924539632,2304,7903340096,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2021209680,2415488,2023625168,7900918560,9924543728,2048,7903336096,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2023641696,4512,2023646208,7900901584,9924547792,47456,7900953552,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2023648080,34033792,2057681872,7866915072,9924596944,3416192,7904365056,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +2057682416,142256,2057824672,7870191440,9928016112,1952,7870335648,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +2057825312,11888304,2069713616,7858306592,9928020208,2424832,7872619728,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +2069748688,5904,2069754592,7860693520,9930448112,1856,7860701280,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +2069796128,10665632,2080461760,7849990448,9930452208,34503808,7895159888,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +2080493008,6576,2080499584,7884458384,9964957968,140736,7884605696,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2080501376,6864,2080508240,7884593024,9965101264,12747232,7897347120,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +2080508704,16200080,2096708784,7881141312,9977850096,2208,7897343600,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +2096709632,5469088,2102178720,7875675472,9977854192,10374624,7891519184,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +2102194704,3648,2102198352,7886033056,9988231408,1920,7886038624,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +2102204848,2896,2102207744,7886027760,9988235504,2528,7886033184,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2102235472,3760,2102239232,7886000368,9988239600,15701312,7901705440,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +2102284000,6896,2102290896,7901652768,10003943664,5469120,7907128784,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +2102310128,6272,2102316400,7907098496,10009414896,1920,7907106688,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +2102324864,4672,2102329536,7907089456,10009418992,1408,7907095536,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +2102336016,5248,2102341264,7907081824,10009423088,1376,7907088448,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +2102346512,6240,2102352752,7907074432,10009427184,3136,7907083808,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +2102357472,6080,2102363552,7907069776,10009433328,1344,7907077200,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +2102377168,8256,2102385424,7907052000,10009437424,6720,7907066976,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +2102391616,6864,2102398480,7907047136,10009445616,1088,7907055088,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +2102403520,3344960,2105748480,7903701232,10009449712,1600,7907047792,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +2105753712,3840,2105757552,7903696256,10009453808,1472,7903701568,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +2105763920,3536,2105767456,7903690448,10009457904,1312,7903695296,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2105835568,3552,2105839120,7903622880,10009462000,27584,7903654016,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +2105865296,3458736,2109324032,7900168656,10009492688,3491904,7907119296,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +2109346016,3472,2109349488,7903638176,10012987664,2528,7903644176,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2109353424,2412672,2111766096,7901225728,10012991824,2016,7903640416,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2111782944,3744,2111786688,7901209136,10012995824,48352,7901261232,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2111788880,25833712,2137622592,7875423408,10013046000,4232896,7905490016,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +2137623056,11820816,2149443872,7867837360,10017281232,4096,7879662272,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +2149444448,3232,2149447680,7867839696,10017287376,2498976,7870341904,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +2149496496,2304640,2151801136,7867987904,10019789040,1952,7870294496,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +2151852384,3701232,2155553616,7864239488,10019793104,24911776,7892852496,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +2155687008,3934192,2159621200,7885085856,10044707056,12388768,7901408816,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +2165039888,5668176,2170708064,7886394224,10057102288,2363744,7894426144,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +2170723840,242069584,2412793424,7646675648,10059469072,3784768,7892530000,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +2412799152,2153008,2414952160,7648303632,10063255792,3747296,7654203936,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +2414960832,3536,2414964368,7652040000,10067004368,5073216,7657116752,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +2414972144,2704,2414974848,7657105808,10072080656,5698400,7662806912,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +2414986000,3376,2414989376,7662790960,10077780336,242541280,7905335616,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +2415037552,9376,2415046928,7905276032,10320322960,2168832,7907454240,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +2415048112,2720096,2417768208,7904725472,10322493680,1920,7907447488,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +2417778112,5248,2417783360,7904714416,10322497776,2944,7904722608,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +2417788816,7801120,2425589936,7896912096,10322502032,1536,7904714752,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +2425616288,5431392,2431047680,7891458288,10322505968,49696,7896939376,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2431064576,3467776,2434532352,7888024880,10322557232,2751232,7894243888,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +2434563984,3520,2434567504,7890743200,10325310704,1952,7890748672,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +2434608336,11424,2434619760,7890695040,10325314800,7992640,7898699104,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +2434661104,4656,2434665760,7898644528,10333310288,5463872,7904113056,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +2434678064,4243856,2438921920,7899853488,10338775408,3477760,7907575104,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +2438941040,3680,2438944720,7903310112,10342254832,2368,7903316160,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2438948816,2419296,2441368112,7900890816,10342258928,2048,7903312160,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2441384912,5664,2441390576,7900872448,10342263024,46048,7900924160,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2441392720,33746240,2475138960,7867173216,10342312176,3447488,7904366944,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +2475139568,141456,2475281024,7870481008,10345762032,1952,7870624416,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +2475281680,12293504,2487575184,7858190944,10345766128,2423296,7872907744,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +2487613072,7264,2487620336,7860571648,10348191984,1920,7860580832,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +2487661520,10706656,2498368176,7849827904,10348196080,34563424,7895097984,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +2498399600,5312,2498404912,7884357376,10382762288,140576,7884503264,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2498406576,3440,2498410016,7884495632,10382905648,12654624,7897153696,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +2498410816,15788176,2514198992,7881363200,10395562192,2688,7897154064,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +2514199536,5491632,2519691168,7875875152,10395566320,10480992,7891847776,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +2519705616,2976,2519708592,7886341440,10406050032,1920,7886346336,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +2519715168,2944,2519718112,7886335984,10406054096,1088,7886340016,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2519741632,2944,2519744576,7886312080,10406056656,15701088,7902016112,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +2519773408,4144,2519777552,7901982688,10421760240,5465056,7907451888,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +2519790416,3824,2519794240,7907434192,10427228432,1920,7907439936,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +2519800800,2880,2519803680,7907428816,10427232496,1408,7907433104,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +2519807424,2880,2519810304,7907426288,10427236592,1280,7907430448,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +2519814960,3104,2519818064,7907422624,10427240688,3232,7907428960,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +2519822432,2736,2519825168,7907421696,10427246864,1344,7907425776,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +2519833776,3104,2519836880,7907414048,10427250928,4832,7907421984,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +2519842208,19840,2519862048,7907394992,10427257040,1120,7907415952,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +2519866368,3589920,2523456288,7903804944,10427261232,1632,7907396496,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +2523460880,3008,2523463888,7903801376,10427265264,1472,7903805856,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +2523469232,3648,2523472880,7903796480,10427269360,1280,7903801408,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2523542000,3472,2523545472,7903727984,10427273456,25920,7903757376,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +2523570768,3566720,2527137488,7900164640,10427302128,3508896,7907240256,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +2527159872,3264,2527163136,7903649008,10430812144,2464,7903654736,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2527167136,2416960,2529584096,7901232400,10430816496,2016,7903651376,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2529601088,3776,2529604864,7901215728,10430820592,48512,7901268016,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2529607056,25387232,2554994288,7875877504,10430871792,3516544,7904781280,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +2554994816,12074240,2567069056,7867322224,10434391280,4384,7879400848,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +2567069680,3744,2567073424,7867324000,10434397424,2736736,7870064480,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +2567117872,2300880,2569418752,7867717872,10437136624,1888,7870020640,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +2569464176,3672864,2573137040,7864003680,10437140720,25314144,7892990688,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +2573239264,3993840,2577233104,7885223072,10462456176,12035712,7901252624,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +2582592704,5692160,2588284864,7886212496,10474497360,2454048,7894358704,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +2588300928,239518224,2827819152,7649134688,10476953840,3817408,7892470320,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +2827823696,2154608,2829978304,7650795056,10480773360,3902560,7656852224,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +2829986400,3600,2829990000,7654688928,10484678928,5091520,7659784048,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +2829997936,2992,2830000928,7659771344,10489772272,5706880,7665481216,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +2830012576,3152,2830015728,7665466368,10495482096,242566784,7908036304,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +2830050096,6160,2830056256,7907995088,10738051344,2171840,7910173088,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +2830056720,2734512,2832791232,7907435024,10740226256,1984,7910171520,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +2832799776,4880,2832804656,7907425728,10740230384,3104,7907433712,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +2832810096,7807648,2840617744,7899618752,10740236496,1536,7907427936,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +2840638672,5434064,2846072736,7894167792,10740240528,47072,7899648928,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2846088784,3462528,2849551312,7890738464,10740289776,2751712,7896952704,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +2849582400,4832,2849587232,7893456080,10743043312,1888,7893462800,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +2849627280,3936,2849631216,7893416192,10743047408,7940800,7901360928,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +2849670288,2944,2849673232,7901316352,10750989584,5517184,7906836480,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +2849684752,3338896,2853023648,7903485264,10756508912,3821792,7910645952,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +2853043536,3520,2853047056,7907285440,10760332496,2496,7907291456,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2853051120,2736096,2855787216,7904549504,10760336720,2048,7907287648,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2855803920,3024,2855806944,7904533776,10760340720,46816,7904583616,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2855808912,33775376,2889584288,7870804528,10760388816,3812448,7908392352,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +2889584848,143824,2889728672,7874475600,10764204272,1920,7874621344,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +2889729328,12812560,2902541888,7861666448,10764208336,2426976,7876905984,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +2902578064,3920,2902581984,7864055312,10766637296,1888,7864061120,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +2902622992,10288288,2912911280,7853730112,10766641392,34306016,7898324416,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +2912943696,3824,2912947520,7888001968,10800949488,140000,7888145792,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2912949200,3328,2912952528,7888139296,10801091824,12269024,7900411648,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +2912953056,15712128,2928665184,7884697232,10813362416,5312,7900414672,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +2928665776,5466656,2934132432,7879238176,10813370608,10774496,7895479328,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +2934147376,2928,2934150304,7889996880,10824147184,4960,7890004768,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +2934156816,2976,2934159792,7889993632,10824153424,7264,7890003872,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2934182864,3328,2934186192,7889977376,10824163568,15771104,7905751808,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +2934215312,6400,2934221712,7905714272,10839935984,5466784,7911187456,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +2934240464,3840,2934244304,7911161120,10845405424,1920,7911166880,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +2934250496,3472,2934253968,7911155648,10845409616,1440,7911160560,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +2934257936,4272,2934262208,7911151440,10845413648,3168,7911158880,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +2934266864,5664,2934272528,7911147232,10845419760,3168,7911156064,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +2934276992,4288,2934281280,7911144528,10845425808,1312,7911150128,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +2934293232,5344,2934298576,7911131424,10845430000,4544,7911141312,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +2934304944,5024,2934309968,7911126176,10845436144,1088,7911132288,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +2934314352,3390096,2937704448,7907735824,10845440272,1632,7911127552,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +2937709184,3520,2937712704,7907731632,10845444336,1472,7907736624,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +2937718192,3728,2937721920,7907726480,10845448400,1312,7907731520,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2937789344,3808,2937793152,7907659376,10845452528,26560,7907689744,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +2937818272,3400384,2941218656,7904262544,10845481200,3503456,7911166384,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +2941239440,3504,2941242944,7907744464,10848987408,2368,7907750336,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2941246800,2415072,2943661872,7905329600,10848991472,2016,7907746688,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +2943678448,4080,2943682528,7905313040,10848995568,48320,7905365440,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +2943684752,24855152,2968539904,7880505808,10849045712,3440576,7908801536,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +2968540400,12476560,2981016960,7871470608,10852487568,4032,7883951200,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +2981017792,3344,2981021136,7871472448,10852493584,2674176,7874149968,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +2981071056,2291840,2983362896,7871807392,10855170288,1920,7874101152,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +2983408976,3667232,2987076208,7868098208,10855174416,25492928,7897258368,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +2987171408,3637776,2990809184,7889859728,10880668912,11806688,7905304192,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +2996262768,5728256,3001991024,7890487904,10892478928,2395648,7898611808,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +3002006256,240199456,3242205712,7652670176,10894875888,3943904,7896813536,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +3242210576,2176672,3244387248,7654434080,10898821328,3916544,7660527296,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +3244395360,3216,3244398576,7658341600,10902740176,5236320,7663581136,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +3244406384,3088,3244409472,7663568464,10907977936,5662944,7669234496,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +3244420688,3776,3244424464,7669218272,10913642736,242114368,7911336416,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +3244467104,5552,3244472656,7911286752,11155759408,2167616,7913459920,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +3244473312,2875520,3247348832,7910579504,11157928336,1920,7913456944,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +3247358544,4768,3247363312,7910569088,11157932400,2912,7910576768,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +3247368800,8248896,3255617696,7902320752,11157938448,1504,7910571152,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +3255642128,5584384,3261226512,7896715872,11157942384,47264,7902347520,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +3261242768,3517104,3264759872,7893231760,11157991632,2756704,7899505568,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +3264791280,4240,3264795520,7895955792,11160751312,1984,7895962016,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +3264837248,4128,3264841376,7895914096,11160755472,7890016,7903808240,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +3264880768,3056,3264883824,7903763648,11168647472,5505440,7909272144,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +3264895408,3440192,3268335600,7905818912,11174154512,3673952,7912933056,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +3268354400,3568,3268357968,7909472672,11177830640,2592,7909478832,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +3268362096,2397056,3270759152,7907075552,11177834704,2048,7909474656,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +3270775536,3584,3270779120,7907062752,11177841872,55936,7907122272,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +3270781040,34240064,3305021104,7872878112,11177899216,4036064,7911154240,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +3305021632,143952,3305165584,7876772288,11181937872,1984,7876918224,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +3305166720,12820208,3317986928,7863955072,11181942000,2423648,7879198928,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +3318023456,4704,3318028160,7866339696,11184367856,1888,7866346288,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +3318069632,10271136,3328340768,7856031184,11184371952,34814112,7901116432,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +3328373232,4432,3328377664,7890810352,11219188016,140544,7890955328,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +3328379472,4208,3328383680,7890947632,11219331312,11914592,7902866432,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +3328384240,15700800,3344085040,7887163552,11231248592,1952,7902866304,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +3344085840,5454672,3349540512,7881712208,11231252720,10789568,7897956448,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +3349554960,3584,3349558544,7892485088,11242043632,2112,7892490784,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +3349565104,3088,3349568192,7892479504,11242047696,960,7892483552,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +3349591824,3424,3349595248,7892454720,11242049968,16312800,7908770944,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +3349623040,4640,3349627680,7908737488,11258365168,5469024,7914211152,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +3349646640,6560,3349653200,7914182272,11263835472,2144,7914190976,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +3349659152,3408,3349662560,7914177936,11263840496,1408,7914182752,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +3349666688,4032,3349670720,7914173872,11263844592,1280,7914179184,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +3349676000,9264,3349685264,7914163456,11263848720,3296,7914176016,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +3349690080,4416,3349694496,7914160368,11263854864,1440,7914166224,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +3349706848,6080,3349712928,7914146000,11263858928,4320,7914156400,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +3349720000,3776,3349723776,7914141296,11263865072,1120,7914146192,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +3349728608,3384080,3353112688,7910756480,11263869168,1600,7914142160,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +3353117456,3856,3353121312,7910751952,11263873264,1504,7910757312,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +3353126912,3968,3353130880,7910746480,11263877360,1248,7910751696,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +3353199920,3456,3353203376,7910678048,11263881424,25120,7910706624,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +3353229040,3450912,3356679952,7907228128,11263908080,3509536,7914188576,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +3356698576,3504,3356702080,7910717264,11267419344,2432,7910723200,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +3356706080,2420976,3359127056,7908296416,11267423472,1984,7910719376,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +3359144912,3536,3359148448,7908279120,11267427568,48000,7908330656,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +3359150528,24892992,3384043520,7883433424,11267476944,3441024,7911767440,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +3384044016,12199824,3396243840,7874676592,11270920432,3776,7886880192,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +3396244448,4320,3396248768,7874677808,11270926576,2436416,7877118544,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +3396289552,2466112,3398755664,7874609056,11273364720,1856,7877077024,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +3398800160,3738544,3402538704,7870830112,11273368816,25923936,7900492592,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +3402629616,3678928,3406308544,7892986928,11299295472,11797920,7908463776,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +3411393344,5711936,3417105280,7893990704,11311095984,2385056,7902087696,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +3417120208,240089808,3657210016,7656274000,11313484016,3934176,7900297984,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +3657213696,2164704,3659378400,7658042896,11317421296,3955424,7664163024,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +3659386144,3040,3659389184,7661990352,11321379536,5190528,7667183920,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +3659396960,2752,3659399712,7667172016,11326571728,5673920,7672848688,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +3659410928,3568,3659414496,7672833296,11332247792,242693856,7915530720,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +3659450432,6176,3659456608,7915487408,11574944016,2174560,7917668144,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +3659457056,2746608,3662203664,7914917312,11577120976,1984,7917665904,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +3662211552,4016,3662215568,7914909536,11577125104,2752,7914916304,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +3662221200,8087248,3670308448,7906820720,11577129168,1536,7914909504,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +3670332848,5503296,3675836144,7901297120,11577133264,45664,7906846080,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +3675853648,3804288,3679657936,7897522432,11577180368,2752608,7904079328,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +3679689168,5072,3679694240,7900241712,11579935952,1920,7900248704,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +3679736720,4640,3679741360,7900198688,11579940048,7985536,7908188864,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +3679780544,2752,3679783296,7908145072,11587928368,5481344,7913629168,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +3679794624,3727696,3683522320,7909888896,11593411216,3706112,7917322704,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +3683541008,3632,3683544640,7913575088,11597119728,2560,7913581280,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +3683548608,2423984,3685972592,7911151264,11597123856,2336,7913577584,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +3685988768,3808,3685992576,7911135344,11597127920,47072,7911186224,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +3685994432,34967904,3720962336,7876215760,11597178096,4002432,7915186096,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +3720962912,140064,3721102976,7880079984,11601182960,1952,7880222000,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +3721103632,12298432,3733402064,7867784736,11601186800,2423616,7882506784,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +3733435888,3968,3733439856,7870172032,11603611888,1888,7870177888,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +3733481504,10711008,3744192512,7859423472,11603615984,34343840,7904478320,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +3744224464,3200,3744227664,7893734336,11637962000,140544,7893878080,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +3744229264,2976,3744232240,7893873088,11638105328,11934080,7905810144,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +3744232656,15762384,3759995040,7890046032,11650041072,2144,7905810560,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +3759995584,5480448,3765476032,7884569136,11650045168,10771616,7900821200,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +3765491776,3520,3765495296,7895324400,11660819696,2016,7895329936,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +3765501904,2816,3765504720,7895319072,11660823792,1056,7895322944,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +3765538512,3344,3765541856,7895286032,11660827888,16033728,7911323104,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +3765569888,4064,3765573952,7911289776,11676863728,5465600,7916759440,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +3765591072,3968,3765595040,7916735856,11682330896,1952,7916741776,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +3765603472,5520,3765608992,7916726000,11682334992,1440,7916732960,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +3765614896,6048,3765620944,7916716992,11682337936,1312,7916724352,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +3765627056,7904,3765634960,7916706112,11682341072,3296,7916717312,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +3765640384,3440,3765643824,7916703424,11682347248,1312,7916708176,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +3765659920,7776,3765667696,7916683552,11682351248,4576,7916695904,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +3765674848,6608,3765681456,7916676032,11682357488,1088,7916683728,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +3765685936,3358880,3769044816,7913316768,11682361584,1632,7916677280,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +3769050176,3632,3769053808,7913311872,11682365680,1472,7913316976,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +3769061072,3280,3769064352,7913305424,11682369776,1280,7913309984,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +3769136304,3712,3769140016,7913233856,11682373872,26848,7913264416,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +3769165856,3364704,3772530560,7909871984,11682402544,3503616,7916740304,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +3772551552,3536,3772555088,7913353632,11685908720,2496,7913359664,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +3772559088,2490416,3775049504,7910863280,11685912784,1984,7913355680,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +3775066480,3248,3775069728,7910847184,11685916912,47392,7910897824,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +3775071872,24828896,3799900768,7886065296,11685966064,3441888,7914336080,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +3799901280,11832448,3811733728,7877676016,11689409744,3264,7889511728,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +3811734336,5424,3811739760,7877676160,11689415920,2435552,7880117136,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +3811784416,2342080,3814126496,7877726576,11691853072,1856,7880070512,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +3814172640,3902688,3818075328,7873781808,11691857136,26544096,7904228592,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +3818183280,3840704,3822023984,7896380352,11718404336,11803168,7912024224,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +3827260368,5660896,3832921264,7897288928,11730210192,2344192,7905294016,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +3832936320,240313328,4073249648,7659306432,11732556080,3879392,7903499152,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +4073255056,2152864,4075407920,7661030080,11736438000,3889536,7667072480,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +4075416160,3232,4075419392,7664910800,11740330192,5382944,7670296976,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +4075427264,2672,4075429936,7670284480,11745714416,5676192,7675963344,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +4075441104,3552,4075444656,7675947872,11751392528,241273408,7917224832,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +4075479744,9200,4075488944,7917178496,11992667440,2164000,7919351696,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +4075489440,2746432,4078235872,7916598256,11994834128,1952,7919346640,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +4078244656,5488,4078250144,7916588112,11994838256,2848,7916596448,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +4078258080,7811200,4086069280,7908774096,11994843376,1536,7916586832,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +4086093296,5441424,4091534720,7903312624,11994847344,47232,7908801280,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +4091551328,3555632,4095106960,7899789632,11994896592,2758016,7906103280,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +4095136960,4832,4095141792,7902515536,11997657328,2080,7902522448,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +4095182592,3664,4095186256,7902475136,11997661392,7871520,7910350320,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +4095227344,3392,4095230736,7910305312,12005536048,5454944,7915763648,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +4095243456,4229696,4099473152,7911520752,12010993904,3469600,7919220048,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +4099493856,3792,4099497648,7914967616,12014465264,2304,7914973712,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +4099501888,2428768,4101930656,7912538704,12014469360,2080,7914969552,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +4101949888,4288,4101954176,7912519280,12014473456,46176,7912569744,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +4101956112,33892784,4135848896,7878673712,12014522608,3422656,7915989152,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +4135849648,208800,4136058448,7881888416,12017946864,1952,7882099168,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +4136059264,12093008,4148152272,7869798688,12017950960,2425152,7884316848,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +4148192720,4048,4148196768,7872182096,12020378864,1888,7872188032,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +4148240304,10670112,4158910416,7861472576,12020382992,34162048,7906304736,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +4158945024,4144,4158949168,7895598528,12054547696,139200,7895741872,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +4158951024,5744,4158956768,7895731504,12054688272,12332832,7908070080,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +4158958112,15223056,4174181168,7892842944,12067024112,1952,7908067952,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +4174181744,5468592,4179650336,7887377872,12067028208,10806528,7903652992,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +4179667904,3488,4179671392,7898166160,12077837552,1920,7898171568,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +4179678432,3264,4179681696,7898159920,12077841616,3072,7898166256,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +4179717152,4320,4179721472,7898126320,12077847792,15536672,7913667312,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +4179750768,5344,4179756112,7913630912,12093387024,5629920,7919266176,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +4179774608,5312,4179779920,7919239072,12099018992,1920,7919246304,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +4179789488,6352,4179795840,7919227312,12099023152,1408,7919235072,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +4179800672,4768,4179805440,7919221744,12099027184,1472,7919227984,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +4179810976,4736,4179815712,7919215568,12099031280,3360,7919223664,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +4179823728,8208,4179831936,7919204464,12099036400,1312,7919213984,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +4179846576,8448,4179855024,7919185472,12099040496,4448,7919198368,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +4179863792,6400,4179870192,7919176448,12099046640,1184,7919184032,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +4179877024,3349632,4183226656,7915824048,12099050704,1824,7919175504,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +4183232528,4288,4183236816,7915818016,12099054832,1632,7915823936,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +4183244512,5040,4183249552,7915809376,12099058928,1280,7915815696,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +4183326688,3440,4183330128,7915732896,12099063024,25696,7915762032,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +4183356880,3447152,4186804032,7912287664,12099091696,3543520,7919278336,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +4186825936,3488,4186829424,7915807360,12102636784,2464,7915813312,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +4186833568,2415376,4189248944,7913391936,12102640880,2304,7915809616,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +4189265776,3760,4189269536,7913375408,12102644944,63616,7913442784,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +4189271600,26073328,4215344928,7887365552,12102710480,3828128,7917267008,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +4215345536,12098864,4227444400,7879096896,12106541296,3296,7891199056,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +4227445392,4864,4227450256,7879097184,12106547440,2443872,7881545920,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +4227496960,2298288,4229795248,7879198528,12108993776,1888,7881498704,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +4229841008,3687968,4233528976,7875468896,12108997872,24936000,7904092864,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +4233627648,3970032,4237597680,7896337664,12133935344,11814080,7912121776,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +4243030608,5744480,4248775088,7896978208,12145753296,2352896,7905075584,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +4248791040,242388480,4491179520,7656927952,12148107472,3725216,7903041648,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +4491195760,2151120,4493346880,7658487952,12151834832,3740704,7664379776,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +4493356256,3280,4493359536,7662219040,12155578576,5090944,7667313264,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +4493367232,3248,4493370480,7667301504,12160671984,5663328,7672968080,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +4493387248,3792,4493391040,7672946736,12166337776,241822912,7914773440,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +4493462000,9328,4493471328,7914691280,12408162608,2157568,7916858176,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +4493473152,2684160,4496157312,7914165872,12410323184,1952,7916851984,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +4496167824,5088,4496172912,7914154368,12410327280,2976,7914162432,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +4496181040,7913696,4504094736,7906236832,12410331568,1504,7914152032,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +4504123424,5432384,4509555808,7900779408,12410335216,50112,7906261904,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +4509576144,3473344,4513049488,7897337216,12410386704,2750304,7903560864,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +4513081040,5680,4513086720,7900052464,12413139184,1920,7900060064,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +4513130544,4320,4513134864,7900008416,12413143280,7919840,7907932576,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +4513173744,2880,4513176624,7907888352,12421064976,5455904,7913347136,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +4513190224,4091328,4517281552,7909242336,12426523888,3468640,7916802304,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +4517301936,3584,4517305520,7912689728,12429995248,2368,7912695680,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +4517309664,2514256,4519823920,7910175424,12429999344,1984,7912691664,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +4519840912,4208,4519845120,7910158288,12430003408,48128,7910210624,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +4519847008,33516368,4553363376,7876691264,12430054640,3424576,7913632208,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +4553364000,159136,4553523136,7879958832,12433481968,1952,7880119920,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +4553523808,12646384,4566170192,7867315904,12433486096,2426240,7882388528,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +4566207952,4416,4566212368,7869701600,12435913968,1856,7869707872,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +4566256128,10383984,4576640112,7859277952,12435918064,33735648,7903397584,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +4576672448,4336,4576676784,7892980064,12469656848,140768,7893125168,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +4576678800,4368,4576683168,7893115984,12469799152,12803680,7905924032,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +4576684032,15657696,4592341728,7890262608,12482604336,2176,7905922480,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +4592342448,5633152,4597975600,7884632768,12482608368,10379136,7900645056,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +4597989904,3024,4597992928,7894996752,12492989680,1952,7895001728,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +4597999840,2912,4598002752,7894991024,12492993776,864,7894994800,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +4598031824,3056,4598034880,7894961040,12492995920,15888288,7910852384,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +4598061184,3136,4598064320,7910822960,12508887280,5504480,7916330576,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +4598081296,5824,4598087120,7916306208,12514393328,2112,7916314144,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +4598093584,2992,4598096576,7916300848,12514397424,1408,7916305248,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +4598100944,2544,4598103488,7916298032,12514401520,1280,7916301856,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +4598108304,3696,4598112000,7916293616,12514405616,3744,7916301056,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +4598116576,3968,4598120544,7916291184,12514411728,1344,7916296496,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +4598130640,9424,4598140064,7916274480,12514414544,4896,7916288800,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +4598150000,5136,4598155136,7916266768,12514421904,1088,7916272992,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +4598160384,3561408,4601721792,7912704304,12514426096,1632,7916267344,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +4601727440,3792,4601731232,7912698960,12514430192,1760,7912704512,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +4601738416,3520,4601741936,7912692352,12514434288,1248,7912697120,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +4601816720,3664,4601820384,7912617968,12514438352,34208,7912655840,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +4601846720,3805840,4605652560,7908822688,12514475248,3824608,7916453136,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +4605672624,3472,4605676096,7912626864,12518302960,2528,7912632864,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +4605680192,2414656,4608094848,7910212208,12518307056,2016,7912628880,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +4608112192,3792,4608115984,7910195168,12518311152,47136,7910246096,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +4608118208,24990336,4633108544,7885251728,12518360272,3841696,7914083760,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +4633109008,12397152,4645506160,7876697184,12522203344,3968,7889098304,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +4645506720,3888,4645510608,7876698944,12522209552,2437024,7879139856,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +4645554736,2307552,4647862288,7876786400,12524648688,1888,7879095840,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +4647908688,3681936,4651590624,7873062192,12524652816,24963488,7901707616,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +4651699728,3762032,4655461760,7894156144,12549617904,12417312,7910335488,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +4661047072,5710176,4666757248,7895281008,12562038256,2346336,7903337520,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +4666774192,241392784,4908166976,7656220080,12564387056,3717536,7901330400,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +4908174112,2158544,4910332656,7657773568,12568106224,3741536,7663673648,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +4910340992,3536,4910344528,7661504512,12571849040,5525152,7667033200,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +4910352144,2864,4910355008,7667021488,12577376496,5689024,7672713376,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +4910367232,3424,4910370656,7672696208,12583066864,242627456,7915327088,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +4910413088,8464,4910421552,7915274080,12825695632,2167232,7917449776,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +4910422000,2728704,4913150704,7914714624,12827865328,1984,7917445312,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +4913160768,4720,4913165488,7914703968,12827869456,2976,7914711664,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +4913172560,7916720,4921089280,7906784464,12827873744,1504,7914702688,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +4921117184,5426288,4926543472,7901334144,12827877616,47264,7906807696,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +4926560624,3453472,4930014096,7897912640,12827926736,2752640,7904118752,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +4930044928,4496,4930049424,7900631904,12830681328,1952,7900638352,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +4930092096,4160,4930096256,7900589168,12830685424,7922240,7908515568,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +4930135168,2944,4930138112,7908472080,12838610192,5462688,7913937712,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +4930149472,3333648,4933483120,7910591072,12844074192,3470944,7917395664,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +4933504144,3568,4933507712,7914038896,12847546608,2432,7914044896,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +4933511808,2401376,4935913184,7911637520,12847550704,2112,7914041008,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +4935929376,4480,4935933856,7911620944,12847554800,47168,7911672592,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +4935935808,33591248,4969527056,7878077920,12847604976,3418752,7915087920,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +4969527616,142640,4969670256,7881354880,12851025136,1952,7881499472,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +4969670896,12772720,4982443616,7868585616,12851029232,2467008,7883825344,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +4982479424,3920,4982483344,7871014752,12853498096,2080,7871020752,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +4982524816,10275824,4992800640,7860701552,12853502192,34443328,7905420704,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +4992833440,3744,4992837184,7895111376,12887948560,140224,7895255344,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +4992838928,4864,4992843792,7895248096,12888091888,11968896,7907221856,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +4992844512,15738864,5008583376,7891480064,12900063440,2208,7907221136,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +5008584112,5500080,5014084192,7885983344,12900067536,10367680,7901851104,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +5014099904,2976,5014102880,7896333712,12910436592,1952,7896338640,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +5014109104,2928,5014112032,7896328656,12910440688,832,7896332416,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +5014138880,4560,5014143440,7896299360,12910442800,15742528,7912046448,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +5014169856,5792,5014175648,7912012112,12926187760,5472992,7917490896,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +5014195104,4288,5014199392,7917462640,12931662032,1984,7917468912,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +5014206288,6832,5014213120,7917453040,12931666160,1440,7917461312,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +5014217936,4896,5014222832,7917447424,12931670256,3712,7917456032,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +5014228768,7920,5014236688,7917439712,12931676400,3456,7917451088,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +5014241648,5712,5014247360,7917435088,12931682448,1344,7917442144,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +5014260144,7968,5014268112,7917418528,12931686640,4352,7917430848,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +5014275904,3952,5014279856,7917412928,12931692784,1120,7917418000,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +5014285120,3681328,5017966448,7913730432,12931696880,1664,7917413424,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +5017971200,19552,5017990752,7913710224,12931700976,1504,7913731280,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +5017996416,30144,5018026560,7913678576,12931705136,1280,7913710000,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +5018098800,5344,5018104144,7913605024,12931709168,25920,7913636288,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +5018135168,3778832,5021914000,7909823936,12931737936,3556960,7917159728,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +5021935264,3312,5021938576,7913358688,12935297264,2464,7913364464,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +5021942544,2419392,5024361936,7910939424,12935301360,2208,7913361024,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +5024378496,3200,5024381696,7910923760,12935305456,48480,7910975440,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +5024383824,24925392,5049309216,7886046544,12935355760,4213088,7915185024,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +5049309856,12452640,5061762496,7877808976,12939571472,3584,7890265200,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +5061763120,4320,5061767440,7877810144,12939577584,2431296,7880245760,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +5061808784,2304256,5064113040,7877897600,12942010640,1888,7880203744,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +5064163072,3670128,5067833200,7874181504,12942014704,24939456,7902791088,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +5067937424,3624112,5071561536,7895394736,12966956272,12481600,7911500448,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +5077065152,5677552,5082742704,7896699168,12979441872,2345504,7904722224,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +5082758608,240696032,5323454640,7658335296,12981789936,3715584,7902746912,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +5323463456,2159600,5325623056,7659883968,12985507024,3743616,7665787184,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +5325632224,3456,5325635680,7663617168,12989252848,5140192,7668760816,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +5325643632,3312,5325646944,7668748432,12994395376,5886496,7674638240,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +5325659968,4432,5325664400,7674620064,13000284464,243240192,7917864688,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +5325716768,7872,5325724640,7917802832,13243527472,2167936,7919978640,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +5325725424,2729152,5328454576,7917243712,13245698288,1952,7919974816,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +5328466304,5568,5328471872,7917230480,13245702352,2784,7917238832,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +5328479744,8462464,5336942208,7908764272,13245706480,1536,7917228272,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +5336973168,5499136,5342472304,7903238272,13245710576,47040,7908784448,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +5342492176,3465536,5345957712,7899803008,13245760720,2864192,7906132736,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +5345988320,3984,5345992304,7902634624,13248626928,1888,7902640496,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +5346036992,3856,5346040848,7902590176,13248631024,7919136,7910513168,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +5346082384,2992,5346085376,7910466320,13256551696,5460800,7915930112,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +5346104944,3333264,5349438208,7912575600,13262013808,3467264,7919376128,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +5349460272,3504,5349463776,7916019216,13265482992,2336,7916025056,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +5349467888,2395360,5351863248,7913623808,13265487056,2080,7916021248,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +5351880144,4112,5351884256,7913606928,13265491184,46144,7913657184,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +5351887056,34268032,5386155088,7879385248,13265540336,3419168,7917072448,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +5386155776,141552,5386297328,7882665184,13268962512,1984,7882808720,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +5386298912,12727552,5399026464,7869940176,13268966640,2430464,7885098192,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +5399061632,4512,5399066144,7872333520,13271399664,1888,7872339920,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +5399110544,10280832,5409391376,7862012384,13271403760,34480480,7906773696,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +5409425840,4128,5409429968,7896457024,13305886992,140800,7896601952,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +5409431760,3632,5409435392,7896594672,13306030064,11942176,7908540480,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +5409435904,15682784,5425118688,7892855536,13317974224,2464,7908540784,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +5425119232,5460704,5430579936,7887398416,13317978352,10435040,7903294160,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +5430594704,3392,5430598096,7897816864,13328414960,1920,7897822176,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +5430605152,2912,5430608064,7897811120,13328419184,896,7897814928,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +5430637760,2992,5430640752,7897782432,13328423184,15770560,7913555984,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +5430669296,5296,5430674592,7913522288,13344196880,5478400,7919005984,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +5430692608,4752,5430697360,7918980960,13349678320,1920,7918987632,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +5430704592,6064,5430710656,7918971760,13349682416,1440,7918979264,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +5430716704,6592,5430723296,7918963184,13349686480,1344,7918971120,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +5430729232,5712,5430734944,7918955632,13349690576,3424,7918964768,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +5430740016,8624,5430748640,7918948112,13349696752,1344,7918958080,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +5430765264,8048,5430773312,7918927536,13349700848,4288,7918939872,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +5430782160,4480,5430786640,7918920352,13349706992,1088,7918925920,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +5430790880,3363024,5434153904,7915557184,13349711088,1632,7918921840,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +5434159440,3440,5434162880,7915552304,13349715184,1472,7915557216,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +5434173248,3232,5434176480,7915542832,13349719312,1440,7915547504,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +5434249024,3952,5434252976,7915470368,13349723344,25152,7915499472,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +5434278272,4237712,5438515984,7911234016,13349750000,3505472,7918977200,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +5438536720,3536,5438540256,7914716944,13353257200,2432,7914722912,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +5438544240,2422000,5440966240,7912295056,13353261296,1984,7914719040,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +5440982816,4608,5440987424,7912277936,13353265360,47424,7912329968,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +5440990144,24870544,5465860688,7887453824,13353314512,4290496,7916614864,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +5465861488,12359840,5478221328,7879388448,13357609776,5088,7891753376,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +5478221968,3504,5478225472,7879391920,13357617392,2470944,7881866368,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +5478268192,2376608,5480644800,7879446576,13360091376,1920,7881825104,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +5480692928,3666048,5484358976,7875736496,13360095472,24951584,7904354128,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +5484459744,3633120,5488092864,7896955472,13385048336,12361888,7912950480,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +5493161744,5827744,5498989488,7898424448,13397413936,2352704,7906604896,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +5499004528,241787904,5740792432,7658976896,13399769328,3794400,7904559200,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +5740797952,2160816,5742958768,7660607552,13403566320,3743264,7666511632,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +5742967360,3296,5742970656,7664340400,13407311056,5097024,7669440720,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +5742978192,3040,5742981232,7669428448,13412409680,5794272,7675225760,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +5742993344,3616,5742996960,7675209488,13418206448,242996160,7918209264,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +5743037280,5984,5743043264,7918162512,13661205776,2172704,7920341200,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +5743043904,2732480,5745776384,7917604304,13663380688,1984,7920338768,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +5745785808,5056,5745790864,7917593952,13663384816,2752,7917601760,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +5745797776,7804032,5753601808,7909787104,13663388912,1536,7917592672,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +5753628240,5424576,5759052816,7904340192,13663393008,45984,7909810752,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +5759069728,3450272,5762520000,7900920240,13663440240,2770784,7907141296,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +5762553232,3936,5762557168,7903655904,13666213072,2016,7903661856,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +5762599360,4016,5762603376,7903613888,13666217264,7897696,7911515600,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +5762642464,3136,5762645600,7911470768,13674116368,5469888,7916943792,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +5762656896,3334304,5765991200,7913596336,13679587536,3478528,7920409168,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +5766011552,3504,5766015056,7917053088,13683068144,2336,7917058928,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +5766019104,2404016,5768423120,7914649216,13683072336,2176,7917055408,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +5768439936,4720,5768444656,7914631872,13683076528,47104,7914683696,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +5768446528,34209936,5802656464,7880469120,13683125584,3417632,7918096688,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +5802656992,142672,5802799664,7883744928,13686544592,1952,7883889552,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +5802800960,11883568,5814684528,7871864192,13686548720,2435296,7886183056,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +5814721504,4272,5814725776,7874260096,13688985872,1920,7874266288,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +5814767856,10265264,5825033120,7863956784,13688989904,35071936,7909293984,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +5825065312,4000,5825069312,7898994736,13724064048,140832,7899139568,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +5825071040,4032,5825075072,7899131248,13724206320,11934080,7911069360,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +5825075664,15700928,5840776592,7895366592,13736143184,2208,7911069728,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +5840777408,5464496,5846241904,7889905280,13736147184,10473696,7905843472,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +5846256736,3344,5846260080,7900363648,13746623728,1920,7900368912,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +5846266816,3264,5846270080,7900357712,13746627792,864,7900361840,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +5846293536,4256,5846297792,7900332144,13746629936,15780192,7916116592,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +5846323424,5328,5846328752,7916083008,13762411760,5487200,7921575536,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +5846348960,4640,5846353600,7921547920,13767901520,2016,7921554576,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +5846359328,3696,5846363024,7921542592,13767905616,1440,7921547728,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +5846367536,4176,5846371712,7921537904,13767909616,1312,7921543392,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +5846376480,4560,5846381040,7921532768,13767913808,3488,7921540816,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +5846386128,5248,5846391376,7921528480,13767919856,1312,7921535040,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +5846403792,7792,5846411584,7921512368,13767923952,4704,7921524864,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +5846418752,4800,5846423552,7921506640,13767930192,1120,7921512560,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +5846428176,3385728,5849813904,7918120320,13767934224,1632,7921507680,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +5849818608,3216,5849821824,7918116560,13767938384,1536,7918121312,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +5849829808,3296,5849833104,7918109376,13767942480,1312,7918113984,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +5849901344,3584,5849904928,7918041552,13767946480,26112,7918071248,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +5849930176,3383456,5853313632,7914661520,13767975152,3522240,7921567216,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +5853333744,3632,5853337376,7918162352,13771499728,2336,7918168320,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +5853341488,2802640,5856144128,7915359728,13771503856,1984,7918164352,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +5856160656,4048,5856164704,7915343248,13771507952,47360,7915394656,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +5856166800,25350048,5881516848,7890040640,13771557488,3504160,7918894848,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +5881517440,11849744,5893367184,7881696128,13775063312,3200,7893549072,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +5893367776,4416,5893372192,7881697264,13775069456,2750464,7884452144,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +5893412624,2454832,5895867456,7881954672,13777822128,1888,7884411392,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +5895914592,3785808,5899700400,7878125632,13777826032,25831808,7907743248,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +5899793728,3810016,5903603744,7900055792,13803659536,12180320,7916046128,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +5908824880,5683632,5914508512,7901334544,13815843056,2370304,7909388480,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +5914523360,241787536,6156310896,7661904768,13818215664,3851904,7907544208,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +6156321488,2161648,6158483136,7663586864,13822070000,3824896,7669573408,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +6158492016,3632,6158495648,7667402032,13825897680,5097152,7672502816,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +6158503184,2768,6158505952,7672490256,13830996208,5708416,7678201440,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +6158519168,3440,6158522608,7678184544,13836707152,243821344,7922009328,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +6158575856,6992,6158582848,7921947856,14080530704,2179040,7924133888,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +6158583328,2722128,6161305456,7921406304,14082711760,1920,7924130352,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +6161317328,5680,6161323008,7921392976,14082715984,2752,7921401408,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +6161330256,7891632,6169221888,7913498224,14082720112,1632,7921391488,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +6169252832,5433936,6174686768,7908037312,14082724080,49440,7913520688,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +6174705776,3459968,6178165744,7904609536,14082775280,2768608,7910838112,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +6178198160,5280,6178203440,7907341760,14085545200,1888,7907348928,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +6178250704,3232,6178253936,7907295360,14085549296,8041856,7915340448,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +6178300896,2896,6178303792,7915289088,14093592880,5523488,7920815472,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +6178317616,3333600,6181651216,7917468096,14099119312,3508768,7924310464,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +6181672848,3808,6181676656,7920953056,14102629712,2336,7920959200,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +6181680736,2402320,6184083056,7918550656,14102633712,2016,7920954992,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +6184102592,3936,6184106528,7918531408,14102637936,45984,7918581328,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +6184108640,34871584,6218980224,7883705712,14102685936,3415424,7921992720,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +6218980960,143328,6219124288,7886979728,14106104016,1952,7887125008,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +6219125776,11922048,6231047824,7875060320,14106108144,2435424,7889417792,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +6231086784,6128,6231092912,7877453376,14108546288,1888,7877461392,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +6231138784,10682672,6241821456,7866728928,14108550384,34749824,7912161424,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +6241855760,4672,6241860432,7901442496,14143302928,139584,7901586752,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +6241862224,4896,6241867120,7901577056,14143444176,12637024,7914218976,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +6241868016,16289184,6258157200,7897926240,14156083440,2208,7914217632,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +6258157968,5467376,6263625344,7892462192,14156087536,10553792,7908483360,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +6263640080,3472,6263643552,7903000400,14166643952,1920,7903005792,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +6263650832,3136,6263653968,7902994080,14166648048,2944,7903000160,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +6263684368,3520,6263687888,7902964384,14166652272,15802976,7918770880,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +6263715600,5712,6263721312,7918735216,14182456528,5484832,7924225760,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +6263739840,4832,6263744672,7924198480,14187943152,1952,7924205264,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +6263751232,6320,6263757552,7924189696,14187947248,1440,7924197456,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +6263763392,4016,6263767408,7924184032,14187951440,1632,7924189680,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +6263772704,5136,6263777840,7924177600,14187955440,3232,7924185968,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +6263783808,3840,6263787648,7924174096,14187961744,1376,7924179312,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +6263798496,9376,6263807872,7924157904,14187965776,4896,7924172176,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +6263817712,4112,6263821824,7924150160,14187971984,1344,7924155616,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +6263826240,3377216,6267203456,7920772560,14187976016,1632,7924151408,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +6267209088,3664,6267212752,7920767232,14187979984,1504,7920772400,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +6267218640,4224,6267222864,7920761248,14187984112,1280,7920766752,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +6267296848,3312,6267300160,7920688272,14187988432,28864,7920720448,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +6267327648,3458704,6270786352,7917232576,14188018928,3517408,7924208688,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +6270807888,3360,6270811248,7920727264,14191538512,2496,7920733120,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +6270815232,2419168,6273234400,7918308112,14191542512,1952,7920729232,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +6273251424,3968,6273255392,7918291312,14191546704,49504,7918344784,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +6273257600,25812640,6299070240,7892528688,14191598928,3880416,7922221744,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +6299070752,11813744,6310884496,7884596320,14195480816,3968,7896414032,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +6310885152,3616,6310888768,7884598192,14195486960,2623360,7887225168,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +6310931472,2333712,6313265184,7884847408,14198112592,1888,7887183008,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +6313309840,3870912,6317180752,7880935840,14198116592,25732096,7910538848,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +6317275360,3877552,6321152912,7902697824,14223850736,12178304,7918753680,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +6326395024,5665520,6332060544,7903971952,14236032496,2502656,7912140128,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +6332076240,240197056,6572273296,7666264448,14238537744,3782816,7910244320,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +6572279072,2168256,6574447328,7667875312,14242322640,3773120,7673816688,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +6574455952,3248,6574459200,7671637904,14246097104,5226624,7676867776,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +6574466752,2752,6574469504,7676856176,14251325680,5718816,7682577744,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +6574481376,3120,6574484496,7682561312,14257045808,245704832,7928269264,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +6574522064,10048,6574532112,7928220544,14502752656,2174368,7930404960,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +6574532720,2738048,6577270768,7927658720,14504929488,1984,7930398752,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +6577279504,5376,6577284880,7927648736,14504933616,2944,7927657056,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +6577291696,7802224,6585093920,7919843952,14504937872,1536,7927647712,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +6585118576,5435712,6590554288,7914387520,14504941808,47296,7919870528,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +6590571168,3462848,6594034016,7910956912,14504990928,2773440,7917193200,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +6594063712,4176,6594067888,7913698208,14507766096,1952,7913704336,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +6594110608,5904,6594116512,7913653712,14507770224,8355648,7922015264,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +6594154912,3600,6594158512,7921970624,14516129136,5595744,7927569968,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +6594169920,3350304,6597520224,7924207024,14521727248,3582016,7931139344,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +6597539136,3600,6597542736,7927768448,14525311184,2336,7927774384,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +6597546848,2405184,6599952032,7925363280,14525315312,2016,7927770480,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +6599969152,4560,6599973712,7925345792,14525319504,45600,7925395952,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +6599975584,33829888,6633805472,7891561072,14525366544,3429056,7928820016,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +6633806032,175568,6633981600,7894816432,14528798032,1984,7894993984,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +6633982224,12457280,6646439504,7882362528,14528802032,2428384,7897248192,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +6646481008,5088,6646486096,7884745888,14531231984,1984,7884752960,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +6646528112,10516608,6657044720,7874191392,14531236112,34769312,7919477312,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +6657077248,3440,6657080688,7908926432,14566007120,140896,7909070768,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +6657082320,7184,6657089504,7909060976,14566150480,12833760,7921901920,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +6657089936,15362736,6672452672,7906534544,14578987216,1952,7921899232,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +6672453664,5615440,6678069104,7900922240,14578991344,10459904,7916997584,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +6678083760,3360,6678087120,7911365472,14589452592,1952,7911370784,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +6678095232,3056,6678098288,7911358336,14589456624,864,7911362256,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +6678121552,2880,6678124432,7911334688,14589459120,15787264,7927124832,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +6678150000,3376,6678153376,7927095376,14605248752,5485600,7932584352,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +6678165232,3792,6678169024,7932567472,14610736496,1952,7932573216,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +6678174992,3008,6678178000,7932562464,14610740464,1472,7932566944,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +6678181984,3264,6678185248,7932559408,14610744656,1312,7932563984,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +6678189792,4400,6678194192,7932554560,14610748752,3552,7932562512,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +6678198608,3456,6678202064,7932552736,14610754800,1344,7932557536,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +6678211984,5968,6678217952,7932541040,14610758992,4800,7932551808,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +6678223552,3360,6678226912,7932538160,14610765072,1120,7932542640,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +6678231568,3575024,6681806592,7928962640,14610769232,1632,7932539296,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +6681811456,3120,6681814576,7928958752,14610773328,1472,7928963344,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +6681819888,4432,6681824320,7928953008,14610777328,1376,7928958816,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +6681888832,5104,6681893936,7928887584,14610781520,27456,7928920144,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +6681918944,3786976,6685705920,7925104336,14610810256,3513824,7932405136,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +6685728432,4176,6685732608,7928594000,14614326608,2656,7928600832,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +6685736672,2417456,6688154128,7926176480,14614330608,1952,7928595888,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +6688170768,4112,6688174880,7926159920,14614334800,54208,7926218240,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +6688177056,26000224,6714177280,7900214768,14614392048,4344608,7930559600,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +6714178192,12033536,6726211728,7892526240,14618737968,4000,7904563776,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +6726212416,5840,6726218256,7892525888,14618744144,2452704,7894984432,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +6726263280,2307232,6728570512,7892628096,14621198608,1856,7894937184,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +6728619216,3665136,6732284352,7888918320,14621202672,25020704,7917604160,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +6732387152,3863344,6736250496,7909974608,14646225104,12401824,7926239776,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +6741690912,5725680,6747416592,7911213216,14658629808,2358784,7919297680,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +6747431696,239621136,6987052832,7673938384,14660991216,3720512,7917280032,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +6987056704,2159792,6989216496,7675497056,14664713552,3752288,7681409136,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +6989224736,3392,6989228128,7679239280,14668467408,5166208,7684408880,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +6989235808,2688,6989238496,7684398192,14673636688,5912992,7690313872,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +6989248816,3344,6989252160,7690300080,14679552240,243265760,7933569184,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +6989288528,5952,6989294480,7933525408,14922819888,2181408,7935712768,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +6989294992,2741440,6992036432,7932966592,14925003024,1984,7935710016,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +6992044784,4256,6992049040,7932958048,14925007088,2944,7932965248,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +6992056096,7828240,6999884336,7925127008,14925011344,1504,7932956752,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +6999907424,5439088,7005346512,7919668768,14925015280,45504,7925153360,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7005362304,3476224,7008838528,7916224848,14925063376,2772544,7922473616,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +7008867744,4288,7008872032,7918966416,14927838448,1888,7918972592,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7008912544,3632,7008916176,7918926336,14927842512,7908448,7926838416,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +7008956144,2752,7008958896,7926794208,14935753104,5470688,7932267648,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +7008970896,3336768,7012307664,7928918560,14941226224,3486752,7935742080,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +7012328992,3440,7012332432,7932383680,14944716112,2336,7932389456,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7012336384,2645664,7014982048,7929738064,14944720112,1952,7932385680,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7014998224,3168,7015001392,7929722912,14944724304,46944,7929773024,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7015003200,34067136,7049070336,7895704240,14944774576,3429888,7933201264,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +7049070928,140496,7049211424,7898994416,14948205840,1952,7899136864,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +7049212080,12770800,7061982880,7886226992,14948209872,2433824,7901431616,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +7062019248,4304,7062023552,7888621456,14950645008,1856,7888627616,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7062063792,10292352,7072356144,7878293024,14950649168,34601024,7923186400,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +7072387568,4080,7072391648,7912860496,14985252144,140448,7913005024,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7072393312,4112,7072397424,7912996992,14985394416,12787328,7925788432,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +7072398064,15671552,7088069616,7910113504,14998183120,1952,7925787008,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +7088070432,5512912,7093583344,7904603904,14998187248,10507648,7920624464,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +7093596752,3088,7093599840,7915097744,15008697584,1920,7915102752,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7093606128,2880,7093609008,7915092800,15008701808,832,7915096512,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7093631696,4848,7093636544,7915067664,15008704208,15794528,7930867040,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +7093661840,5856,7093667696,7930833280,15024500976,5490432,7936329568,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +7093686384,7424,7093693808,7936298912,15029992720,2080,7936308416,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +7093699696,3312,7093703008,7936293808,15029996816,1440,7936298560,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +7093706912,5824,7093712736,7936288176,15030000912,3392,7936297392,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +7093717552,5248,7093722800,7936284224,15030007024,3488,7936292960,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +7093727600,4640,7093732240,7936280928,15030013168,1312,7936286880,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +7093742736,5232,7093747968,7936269296,15030017264,4608,7936279136,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +7093753216,4880,7093758096,7936265280,15030023376,1120,7936271280,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +7093762704,3639856,7097402560,7932624944,15030027504,1664,7936266464,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +7097407264,9408,7097416672,7932614928,15030031600,1472,7932625808,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +7097422384,10256,7097432640,7932603056,15030035696,1280,7932614592,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7097497648,36288,7097533936,7932505856,15030039792,26944,7932569088,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +7097558368,3947872,7101506240,7928562224,15030068464,3510464,7936020560,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +7101528944,3520,7101532464,7932048288,15033580752,2432,7932054240,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7101536592,2415824,7103952416,7929632432,15033584848,2016,7932050272,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7103970496,4240,7103974736,7929614208,15033588944,47936,7929666384,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7103979248,24840160,7128819408,7904819744,15033639152,4327456,7933987360,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +7128820160,12402896,7141223056,7896745600,15037968656,3584,7909152080,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +7141223824,4192,7141228016,7896746784,15037974800,2438464,7899189440,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +7141276752,2292016,7143568768,7896847216,15040415984,1856,7899141088,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7143623120,3653136,7147276256,7893143824,15040420080,24980000,7921776960,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +7147384720,3633232,7151017952,7914383760,15065401712,12355584,7930372576,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +7156533776,5683504,7162217280,7915544240,15077761520,2391040,7923618784,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +7162232272,242216400,7404448672,7675706800,15080155472,3754496,7921677696,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +7404462752,2165984,7406628736,7677283664,15083912400,3751616,7683201264,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +7406639280,3456,7406642736,7681022656,15087665392,5103328,7686129440,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +7406651376,3200,7406654576,7686115424,15092770000,5872896,7691991520,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +7406674176,3472,7406677648,7691967168,15098644816,243853824,7935824464,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +7406792976,12672,7406805648,7935694464,15342500112,2173248,7937880384,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +7406806112,2635568,7409441680,7935233344,15344675024,1920,7937870832,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +7409455984,5824,7409461808,7935217344,15344679152,2944,7935226112,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +7409472464,7827936,7417300400,7927382976,15344683376,1504,7935212416,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +7417356112,5407488,7422763600,7921923744,15344687344,45984,7927377216,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7422789296,3443936,7426233232,7918501408,15344734640,2776768,7924722112,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +7426270432,4496,7426274928,7921237792,15347512720,1888,7921244176,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7426330976,3552,7426334528,7921182128,15347516656,7939712,7929125392,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +7426389472,3984,7426393456,7929065408,15355458864,5468864,7934538256,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +7426407904,3291648,7429699552,7931229456,15360929008,3482720,7938003824,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +7429720096,3600,7429723696,7934690016,15364413712,2432,7934696048,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7429727952,2401216,7432129168,7932288608,15364417776,2144,7934691968,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7432147600,3952,7432151552,7932270320,15364421872,46048,7932320320,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7432154192,34858752,7467012944,7897457056,15364470000,3435008,7935750816,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +7467013632,144624,7467158256,7900748288,15367906544,1920,7900894832,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +7467159168,11881472,7479040640,7888870000,15367910640,2439200,7903190672,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +7479081632,4208,7479085840,7891266016,15370351856,1920,7891272144,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7479132288,10274656,7489406944,7880949008,15370355952,34622048,7925845712,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +7489441600,4000,7489445600,7915534928,15404980528,142464,7915681392,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7489447312,5312,7489452624,7915672320,15405124944,12814048,7928491680,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +7489453088,15689456,7505142544,7912798688,15417941232,1984,7928490128,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +7505143104,5524048,7510667152,7907278176,15417945328,10463552,7923265776,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +7510683168,3072,7510686240,7917724464,15428410704,1952,7917729488,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7510694448,3184,7510697632,7917717168,15428414800,864,7917721216,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7510737328,4672,7510742000,7917676832,15428418832,15803776,7933485280,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +7510774336,5184,7510779520,7933445744,15444225264,5480064,7938930992,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +7510797568,9520,7510807088,7938900768,15449707856,1920,7938912208,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +7510818192,6000,7510824192,7938887760,15449711952,1472,7938895232,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +7510834160,5776,7510839936,7938876016,15449715952,1312,7938883104,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +7510845088,6992,7510852080,7938868064,15449720144,3552,7938878608,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +7510858656,5680,7510864336,7938861856,15449726192,1344,7938868880,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +7510891344,8096,7510899440,7938830848,15449730288,4480,7938843424,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +7510912320,8656,7510920976,7938815616,15449736592,1088,7938825360,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +7510926112,3347008,7514273120,7935467376,15449740496,1632,7938816016,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +7514278400,3488,7514281888,7935462960,15449744848,1504,7935467952,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +7514289168,4960,7514294128,7935454688,15449748816,1280,7935460928,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7514381488,3344,7514384832,7935367984,15449752816,25952,7935397280,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +7514411104,4188768,7518599872,7931181616,15449781488,3512480,7938882864,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +7518622256,3328,7518625584,7934670368,15453295952,2400,7934676096,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7518629680,2424736,7521054416,7932245504,15453299920,2176,7934672416,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7521071216,4288,7521075504,7932228640,15453304144,47104,7932280032,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7521078000,25207904,7546285904,7907068352,15453354256,4206304,7936482560,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +7546286944,12455456,7558742400,7898820496,15457562896,3264,7911279216,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +7558743184,4384,7558747568,7898821472,15457569040,2482784,7901308640,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +7558793792,2305520,7561099312,7898955040,15460054352,1888,7901262448,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7561151488,3665216,7564816704,7895241744,15460058448,24982208,7923889168,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +7564912816,3691568,7568604384,7916437552,15485041936,12379232,7932508352,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +7573762864,5813600,7579576464,7917848864,15497425328,2372352,7926034816,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +7579591184,240910720,7820501904,7679297888,15499799792,3764256,7923972864,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +7820506720,2152560,7822659280,7680906752,15503566032,3745824,7686805136,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +7822667424,3248,7822670672,7684644224,15507314896,5102048,7689749520,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +7822678816,3232,7822682048,7689736592,15512418640,5740512,7695480336,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +7822694960,3328,7822698288,7695463616,15518161904,243390656,7938857600,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +7822740064,10224,7822750288,7938803584,15761553872,2179552,7940993360,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +7822750768,2749296,7825500064,7938235696,15763735760,1984,7940986976,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +7825508544,4608,7825513152,7938226736,15763739888,2848,7938234192,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +7825519424,7836528,7833355952,7930388096,15763744048,1504,7938226128,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +7833387808,5421408,7838809216,7924938832,15763748048,46464,7930406704,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7838825824,3462656,7842288480,7921507824,15763796304,2774592,7927745072,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +7842318608,3872,7842322480,7924250816,15766573296,1920,7924256608,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7842364000,3024,7842367024,7924210368,15766577392,7937408,7932150800,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +7842406240,3232,7842409472,7932108080,15774517552,5466464,7937577776,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +7842421040,3344848,7845765888,7934219760,15779985648,3486240,7941050848,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +7845786992,3920,7845790912,7937683600,15783474512,2400,7937689920,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7845795104,2401632,7848196736,7935281872,15783478608,2016,7937685520,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7848213376,4752,7848218128,7935264480,15783482608,45856,7935315088,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7848220272,34252944,7882473216,7901057648,15783530864,3417632,7938728224,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +7882473856,141408,7882615264,7904334576,15786949840,1952,7904477936,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +7882616224,11866480,7894482704,7892471264,15786953968,2433792,7906771536,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +7894523184,5408,7894528592,7894861472,15789390064,1856,7894868736,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7894571744,10267072,7904838816,7884555344,15789394160,34660288,7929482704,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +7904871072,4368,7904875440,7919182176,15824057616,140576,7919327120,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7904877040,2960,7904880000,7919320944,15824200944,12779424,7932103328,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +7904880432,15702960,7920583392,7916399088,15836982480,2496,7932104544,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +7920583920,5465408,7926049328,7910937280,15836986608,10441280,7926843968,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +7926063744,3600,7926067344,7921362112,15847429456,2016,7921367728,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7926074176,3344,7926077520,7921356032,15847433552,2912,7921362288,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7926107760,4560,7926112320,7921327280,15847439600,15823232,7937155072,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +7926141568,6848,7926148416,7937116080,15863264496,5483136,7942606064,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +7926164704,4656,7926169360,7942580832,15868750192,1920,7942587408,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +7926176912,5136,7926182048,7942572112,15868754160,1440,7942578688,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +7926187376,6320,7926193696,7942564656,15868758352,3424,7942574400,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +7926199664,5360,7926205024,7942559344,15868764368,3296,7942568000,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +7926209904,5856,7926215760,7942554880,15868770640,1344,7942562080,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +7926229792,7264,7926237056,7942537712,15868774768,4416,7942549392,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +7926244224,5024,7926249248,7942531600,15868780848,1088,7942537712,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +7926254416,3377264,7929631680,7939153328,15868785008,1632,7942532224,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +7929636672,3216,7929639888,7939149088,15868788976,1504,7939153808,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +7929645696,3840,7929649536,7939143536,15868793072,1280,7939148656,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7929720336,3296,7929723632,7939073664,15868797296,26336,7939103296,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +7929749328,3391392,7933140720,7935685120,15868825840,3516480,7942592992,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +7933163216,3584,7933166800,7939178528,15872345328,2336,7939184448,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7933170800,2789456,7935960256,7936389168,15872349424,2016,7939180640,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +7935976976,4352,7935981328,7936372192,15872353520,47872,7936424416,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +7935983584,25276112,7961259696,7911143040,15872402736,3900736,7940319888,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +7961260192,11802160,7973062352,7903242880,15876305232,1952,7915046992,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +7973063280,4688,7973067968,7903241264,15876309232,2704032,7905949984,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +7973112496,2335808,7975448304,7903566304,15879014608,2208,7905904320,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +7975495376,3880560,7979375936,7899642800,15879018736,25014784,7928538144,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +7979476720,3819504,7983296224,7920739824,15904036048,12223424,7936782752,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +7988515152,5693456,7994208608,7922053392,15916262000,2486112,7930232960,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +7994223696,241138400,8235362096,7683388864,15918750960,3770784,7928298048,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +8235370144,2176288,8237546432,7684977968,15922524400,3784096,7690938352,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +8237554832,3264,8237558096,7688753056,15926311152,5109984,7693866304,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +8237566352,3024,8237569376,7693854608,15931423984,5723136,7699580768,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +8237581904,3296,8237585200,7699563968,15937149168,243361312,7942928576,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +8237633664,4848,8237638512,7942874528,16180513040,2178944,7945058320,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +8237639888,2733536,8240373424,7942320800,16182694224,1952,7945056288,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +8240384176,4528,8240388704,7942309616,16182698320,3200,7942317344,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +8240397328,7935408,8248332736,7934371632,16182704368,1536,7942308576,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +8248367040,5417088,8253784128,7928924432,16182708560,49120,7934390640,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +8253802752,3455024,8257257776,7925502976,16182760752,2770976,7931728976,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +8257290112,4432,8257294544,7928240192,16185534736,1920,7928246544,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +8257339472,3344,8257342816,7928196080,16185538896,8428352,7936627776,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +8257388256,3888,8257392144,7936576480,16193968624,5499328,7942079696,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +8257404752,3342272,8260747024,7938722400,16199469424,3676128,7945740800,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +8260768720,3616,8260772336,7942375168,16203147504,2496,7942381280,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +8260776544,2402544,8263179088,7939972512,16203151600,2080,7942377136,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +8263195904,4000,8263199904,7939955760,16203155664,47648,7940007408,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +8263201840,34368272,8297570112,7905634768,16203204880,4130240,7944133280,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +8297571328,144064,8297715392,7909621424,16207336816,1952,7909767440,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +8297716032,11915616,8309631648,7897709136,16207340784,2429920,7912054672,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +8309669232,4608,8309673840,7900099936,16209773776,1888,7900106432,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +8309714416,10723856,8320438272,7889339728,16209778000,34611488,7934675072,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +8320471904,3680,8320475584,7923915600,16244391184,139936,7924059216,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +8320477888,4960,8320482848,7924049616,16244532464,12928448,7936983024,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +8320483296,16076352,8336559648,7920903856,16257463504,2208,7936982416,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +8336560496,5466480,8342026976,7915440656,16257467632,10450016,7931357152,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +8342042368,3248,8342045616,7925873984,16267919600,1920,7925879152,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +8342052832,3360,8342056192,7925867600,16267923792,832,7925871792,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +8342086032,3280,8342089312,7925836976,16267926288,15834848,7941675104,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +8342119504,5264,8342124768,7941638224,16283762992,5478816,7947122304,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +8342142512,7440,8342149952,7947094448,16289244400,1952,7947103840,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +8342158736,7728,8342166464,7947082032,16289248496,1408,7947091168,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +8342172144,6608,8342178752,7947072848,16289251600,1472,7947080928,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +8342184496,5200,8342189696,7947064912,16289254608,3456,7947073568,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +8342196320,6192,8342202512,7947058272,16289260784,1376,7947065840,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +8342216496,7984,8342224480,7947040400,16289264880,4800,7947053184,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +8342233840,4896,8342238736,7947032256,16289270992,1088,7947038240,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +8342243904,3351936,8345595840,7943679280,16289275120,1600,7947032816,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +8345600784,3824,8345604608,7943674608,16289279216,1504,7943679936,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +8345610944,4336,8345615280,7943668032,16289283312,1280,7943673648,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +8345686176,3408,8345689584,7943597824,16289287408,26624,7943627856,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +8345715600,3385056,8349100656,7940215424,16289316080,3510240,7947110720,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +8349121792,3408,8349125200,7943704320,16292829520,2432,7943710160,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +8349129184,2427120,8351556304,7941277184,16292833488,2048,7943706352,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +8351573168,4368,8351577536,7941260080,16292837616,48736,7941313184,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +8351579824,25721168,8377300992,7915586768,16292887760,4145728,7945453664,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +8377301472,11895872,8389197344,7907838704,16297036048,3808,7919738384,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +8389197968,4896,8389202864,7907839328,16297042192,2516704,7910360928,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +8389252016,2295872,8391547888,7908012320,16299560208,1888,7910310080,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +8391595104,3716848,8395311952,7904252320,16299564272,25347424,7933316592,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +8395413040,3928512,8399341552,7925572864,16324914416,12396480,7941897856,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +8404761968,5662512,8410424480,7926890032,16337314512,2354336,7934906880,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +8410439184,241582832,8652022016,7687649264,16339671280,3785536,7933017632,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +8652028000,2158448,8654186448,7689272608,16343459056,3750080,7695181136,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +8654194800,4688,8654199488,7693012464,16347211952,5090848,7698108000,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +8654207312,3184,8654210496,7698093872,16352304368,5793280,7703890336,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +8654222832,3456,8654226288,7703872960,16358099248,242750816,7946627232,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +8654273888,5520,8654279408,7946572736,16600852144,2176192,7948754448,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +8654280384,2722736,8657003120,7946027744,16603030864,1984,7948752464,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +8657011872,3952,8657015824,7946019040,16603034864,3008,7946026000,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +8657021584,7930096,8664951680,7938089328,16603041008,1536,7946020960,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +8664975728,5437824,8670413552,7932631520,16603045072,48992,7938118336,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +8670429312,3462016,8673891328,7929204944,16603096272,2770880,7935437840,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +8673920864,4016,8673924880,7931944416,16605869296,2016,7931950448,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +8673966160,2928,8673969088,7931904304,16605873392,8377760,7940284992,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +8674009696,2912,8674012608,7940241200,16614253808,5504512,7945748624,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +8674023520,4275808,8678299328,7941460528,16619759856,3685984,7949422320,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +8678317424,3664,8678321088,7945126288,16623447376,2432,7945132384,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +8678325168,2425520,8680750688,7942700656,16623451344,2176,7945128352,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +8680767632,3904,8680771536,7942684032,16623455568,47456,7942735392,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +8680773552,34560976,8715334528,7908170096,16623504624,3658144,7946389216,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +8715335296,142624,8715477920,7911687504,16627165424,1952,7911832080,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +8715479152,12381840,8727860992,7899308528,16627169520,2434048,7914124416,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +8727906576,5472,8727912048,7901693568,16629605616,1920,7901700960,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +8727973776,10337424,8738311200,7891298512,16629609712,34581888,7936217824,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +8738354352,4528,8738358880,7925834416,16664193296,141376,7925980320,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +8738361280,5024,8738366304,7925970288,16664336592,12743040,7938718352,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +8738366992,15264272,8753631264,7923451056,16677082320,2208,7938717536,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +8753632368,5478080,8759110448,7917975968,16677086416,10444640,7933898688,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +8759129936,3248,8759133184,7928399152,16687532336,1952,7928404352,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +8759141696,3280,8759144976,7928391392,16687536368,928,7928395600,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +8759228944,3824,8759232768,7928305808,16687538576,15807712,7944117344,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +8759289824,7104,8759296928,7944052080,16703349008,5580608,7949639792,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +8759350048,13664,8759363712,7949568112,16708931824,1920,7949583696,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +8759388112,8912,8759397024,7949539056,16708936080,1440,7949549408,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +8759409616,6704,8759416320,7949523760,16708940080,1312,7949531776,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +8759427408,13856,8759441264,7949504736,16708946000,22080,7949540672,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +8759448880,9792,8759458672,7949597888,16709056560,1344,7949609024,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +8759519552,14720,8759534272,7949525552,16709059824,5952,7949546224,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +8759566176,13888,8759580064,7949487952,16709068016,1120,7949502960,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +8759586224,3295552,8762881776,7946190464,16709072240,6144,7949492160,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +8762888848,4368,8762893216,7946317488,16709210704,1952,7946323808,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +8762906064,4656,8762910720,7946304752,16709215472,1248,7946310656,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +8763025376,7440,8763032816,7946186720,16709219536,38432,7946232592,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +8763074480,3508112,8766582592,7942676912,16709259504,3551520,7949736544,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +8766605088,3552,8766608640,7946204112,16712812752,5216,7946212880,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +8766612528,2419360,8769031888,7943789056,16712820944,2432,7946210848,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +8769052480,3536,8769056016,7943769056,16712825072,49920,7943822512,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +8769058256,25532576,8794590832,7918286464,16712877296,4215904,7948034944,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +8794591296,12078096,8806669392,7910426752,16717096144,4128,7922508976,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +8806669952,3792,8806673744,7910428576,16717102320,2539872,7912972240,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +8806732176,2291728,8809023904,7910620080,16719643984,1856,7912913664,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +8809081648,3666432,8812748080,7906899872,16719647952,24050080,7934616384,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +8812959120,3891648,8816850768,7926848960,16743699728,12171808,7942912416,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +8822205072,5719472,8827924544,7927950800,16755875344,2411200,7936081472,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +8827943840,241592832,9069536672,7688767760,16758304432,3849120,7934209712,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +9069554896,2155264,9071710160,7690445088,16762155248,3791648,7696392000,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +9071720080,3136,9071723216,7694225920,16765949136,5113664,7699342720,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +9071732128,3104,9071735232,7699329936,16771065168,5721440,7705054480,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +9071757200,4832,9071762032,7705027200,16776789232,243282592,7948314624,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +9071845504,10688,9071856192,7948218160,17020074352,2181056,7950409904,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +9071858192,2674176,9074532368,7947725088,17022257456,1952,7950401216,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +9074542976,6144,9074549120,7947712368,17022261488,2752,7947721264,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +9074559056,7900928,9082459984,7939805696,17022265680,1536,7947708160,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +9082503840,5420400,9087924240,7934344608,17022268848,47200,7939812208,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +9087949744,3463008,9091412752,7930905024,17022317776,2768704,7937136736,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +9091444400,4592,9091448992,7933639056,17025088048,1888,7933645536,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +9091490976,3984,9091494960,7933596864,17025091824,7918304,7941519152,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +9091536800,3008,9091539808,7941472688,17033012496,5477536,7946953232,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +9091552896,3793872,9095346768,7943146240,17038493008,3490304,7950430416,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +9095367760,3648,9095371408,7946613344,17041984752,2592,7946619584,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +9095375568,2602960,9097978528,7944010320,17041988848,1952,7946615232,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +9097995936,5056,9098000992,7943991952,17041992944,47552,7944044560,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +9098002944,33679648,9131682592,7910360496,17042043088,3416192,7947456336,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +9131683136,142512,9131825648,7913635552,17045461200,2048,7913780112,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +9131826304,11921232,9143747536,7901717792,17045465328,2434656,7916073680,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +9143785760,4032,9143789792,7904112784,17047902576,1920,7904118736,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +9143834640,10302048,9154136688,7893769824,17047906512,34610912,7938682784,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +9154172848,4192,9154177040,7928341792,17082518832,140352,7928486336,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +9154178848,5216,9154184064,7928477072,17082661136,11964480,7940446768,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +9154184528,15674288,9169858816,7924769840,17094628656,2208,7940446336,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +9169860304,5507232,9175367536,7919265120,17094632656,10482784,7935255136,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +9175383552,3328,9175386880,7929730896,17105117776,1920,7929736144,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +9175394576,4544,9175399120,7929722432,17105121552,864,7929727840,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +9175444112,3168,9175447280,7929676416,17105123696,15782656,7945462240,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +9175485568,3120,9175488688,7945419840,17120908528,5489472,7950912432,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +9175508704,8800,9175517504,7950882736,17126400240,1984,7950893520,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +9175532896,7136,9175540032,7950864304,17126404336,1440,7950872880,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +9175546720,6784,9175553504,7950855344,17126408848,3520,7950865648,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +9175560832,8544,9175569376,7950845168,17126414544,3456,7950857168,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +9175575952,7424,9175583376,7950837440,17126420816,1344,7950846208,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +9175621088,13728,9175634816,7950790000,17126424816,4448,7950808176,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +9175694400,6576,9175700976,7950730080,17126431056,1120,7950737776,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +9175707424,3588192,9179295616,7947139568,17126435184,1632,7950729392,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +9179301920,5056,9179306976,7947132176,17126439152,1472,7947138704,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +9179315984,4032,9179320016,7947123360,17126443376,1248,7947128640,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +9179415840,3936,9179419776,7947027664,17126447440,26944,7947058544,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +9179454496,3731184,9183185680,7943290400,17126476080,3512928,7950534512,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +9183207680,3312,9183210992,7946780512,17129991504,2400,7946786224,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +9183215088,2403584,9185618672,7944376832,17129995504,2080,7946782496,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +9185636640,4256,9185640896,7944358928,17129999824,47872,7944411056,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +9185643280,24602576,9210245856,7919804272,17130050128,3447936,7947854784,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +9210246512,12468592,9222715104,7910784624,17133499728,3616,7923256832,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +9222715968,3536,9222719504,7910786368,17133505872,2833664,7913623568,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +9222764704,2305392,9225070096,7911271136,17136341232,1888,7913578416,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +9225121152,3679872,9228801024,7907544400,17136345424,25380896,7936605168,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +9228948704,3587632,9232536336,7929192896,17161729232,11852064,7944632592,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +9238067328,5798816,9243866144,7929718480,17173584624,2479616,7937996912,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +9243884080,242194624,9486078704,7689988608,17176067312,3863872,7936047104,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +9486095072,2150672,9488245744,7691687168,17179932912,3950816,7697788656,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +9488255824,3488,9488259312,7695626208,17183885520,5209696,7700839392,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +9488268208,2720,9488270928,7700825856,17189096784,5707104,7706535680,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +9488297024,4944,9488301968,7706503520,17194805488,243530720,7950039184,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +9488395376,9120,9488404496,7949933824,17438338320,2179584,7952122528,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +9488405392,2665456,9491070848,7949449552,17440520400,2048,7952117056,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +9491082384,5840,9491088224,7949436304,17440524528,2752,7949444896,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +9491098944,7923936,9499022880,7941505712,17440528592,1536,7949431184,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +9499066240,5412928,9504479168,7936053520,17440532688,45664,7941512112,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +9504498704,3451984,9507950688,7932630160,17440580848,2772320,7938854464,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +9507981856,4192,9507986048,7935368816,17443354864,1888,7935374896,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +9508032496,3776,9508036272,7935322688,17443358960,7933664,7943260128,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +9508079104,2880,9508081984,7943211952,17451293936,5481440,7948696272,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +9508095536,3336592,9511432128,7945345296,17456777424,3490848,7952172736,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +9511454736,3664,9511458400,7948811920,17460270320,2304,7948817888,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +9511462736,2457488,9513920224,7946354192,17460274416,2176,7948813856,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +9513936960,3600,9513940560,7946337984,17460278544,46048,7946387632,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +9513942480,34361680,9548304160,7912023600,17460327760,3426208,7949811488,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +9548304704,141440,9548446144,7915310864,17463757008,1952,7915454256,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +9548447904,12797392,9561245296,7902515840,17463761136,2438752,7917751984,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +9561282800,3856,9561286656,7904915696,17466202352,1888,7904921440,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +9561328784,10280688,9571609472,7894596976,17466206448,34688000,7939565664,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +9571644000,3888,9571647888,7929248640,17500896528,140192,7929392720,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +9571650112,3968,9571654080,7929385776,17501039856,12220320,7941610064,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +9571654816,15716592,9587371408,7925890144,17513261552,2016,7941608752,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +9587372464,5458352,9592830816,7920434608,17513265424,10986432,7936879392,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +9592846560,3472,9592850032,7931403904,17524253936,1952,7931409328,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +9592857264,3232,9592860496,7931397536,17524258032,896,7931401664,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +9592901568,8208,9592909776,7931350432,17524260208,15775616,7947134256,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +9592938608,2976,9592941584,7947096288,17540037872,5476288,7952575552,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +9592960784,5904,9592966688,7952549584,17545516272,1952,7952557440,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +9592976752,5360,9592982112,7952538256,17545520368,1440,7952545056,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +9592990256,4864,9592995120,7952529344,17545524464,1280,7952535488,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +9593001168,6512,9593007680,7952520880,17545528560,3040,7952530432,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +9593014720,5296,9593020016,7952514688,17545534704,1440,7952521424,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +9593034496,9632,9593044128,7952493264,17545537392,4672,7952507568,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +9593052880,4080,9593056960,7952487056,17545544016,1088,7952492224,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +9593061488,3480256,9596541744,7949006272,17545548016,1600,7952488128,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +9596547472,14992,9596562464,7948989648,17545552112,1504,7949006144,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +9596568416,8112,9596576528,7948979808,17545556336,1248,7948989168,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +9596657552,3232,9596660784,7948899520,17545560304,25984,7948928736,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +9596690480,4077072,9600767552,7944821424,17545588976,3520064,7952418560,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +9600790032,3152,9600793184,7948317328,17549110512,2688,7948323168,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +9600797136,2398464,9603195600,7945919008,17549114608,2016,7948319488,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +9603214272,5344,9603219616,7945899056,17549118672,49440,7945953840,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +9603222352,24909936,9628132288,7921037616,17549169904,3445856,7949393408,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +9628132960,12467968,9640600928,7912016880,17552617808,3680,7924488528,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +9640601840,4032,9640605872,7912017952,17552623824,2677920,7914699904,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +9640649808,2305344,9642955152,7912348608,17555303760,1888,7914655840,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +9643007808,3679280,9646687088,7908620768,17555307856,25633504,7937933552,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +9646813632,3625584,9650439216,7930504416,17580943632,11815744,7945945744,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +9655795904,5783504,9661579408,7931183424,17592762832,2383392,7939350320,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +9661594736,241181248,9902775984,7692371520,17595147504,3935872,7937488640,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +9902783600,2161168,9904944768,7694141040,17599085808,3919136,7700221344,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +9904952944,3200,9904956144,7698050560,17603006704,5242336,7703296096,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +9904964352,2960,9904967312,7703283392,17608250704,5685216,7708971568,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +9904979552,4192,9904983744,7708955184,17613938928,243162464,7952121840,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +9905031312,8000,9905039312,7952063808,17857103120,2174272,7954246080,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +9905039824,2722080,9907761904,7951518240,17859280144,1920,7954242240,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +9907771376,5648,9907777024,7951507184,17859284208,2624,7951515456,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +9907783040,7830512,9915613552,7943674720,17859288272,1536,7951506768,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +9915639008,5440768,9921079776,7938212624,17859292400,45184,7943698576,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +9921096624,3450848,9924547472,7934792000,17859339472,2773504,7941016352,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +9924577200,4256,9924581456,7937532800,17862114256,1920,7937538976,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +9924621872,3040,9924624912,7937493760,17862118672,7919360,7945416160,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +9924664912,2704,9924667616,7945372848,17870040464,5467552,7950843104,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +9924678944,3344800,9928023744,7947486736,17875510480,3489920,7954321456,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +9928041984,3536,9928045520,7950957920,17879003440,2400,7950963856,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +9928049728,2406000,9930455728,7948551808,17879007536,1952,7950959760,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +9930472160,5056,9930477216,7948534352,17879011568,47104,7948586512,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +9930479312,34486272,9964965584,7914096160,17879061744,3419008,7952001440,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +9964966112,146560,9965112672,7917369392,17882482064,1984,7917517936,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +9965114944,12742720,9977857664,7904628432,17882486096,2434304,7919805456,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +9977892880,4832,9977897712,7907025408,17884923120,1920,7907032160,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +9977939488,10299984,9988239472,7896687744,17884927216,34663328,7941651056,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +9988272720,3760,9988276480,7931315408,17919591888,140288,7931459456,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +9988278528,5072,9988283600,7931451424,17919735024,11913600,7943370096,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +9988284160,15667456,10003951616,7927698640,17931650256,2176,7943368272,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +10003952288,5470336,10009422624,7922231760,17931654384,10436992,7938139088,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +10009437376,3328,10009440704,7932653360,17942094064,1920,7932658608,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +10009446992,2944,10009449936,7932648256,17942098192,2720,7932653920,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +10009475088,4896,10009479984,7932622304,17942102288,15837408,7948464608,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +10009506448,4896,10009511344,7948430144,17957941488,5479968,7953915008,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +10009529888,6080,10009535968,7953886992,17963422960,1920,7953894992,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +10009542304,3728,10009546032,7953881120,17963427152,1408,7953886256,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +10009549984,4960,10009554944,7953876304,17963431248,1312,7953882576,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +10009560416,5664,10009566080,7953869136,17963435216,3104,7953877904,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +10009571136,4320,10009575456,7953866032,17963441488,1376,7953871728,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +10009590912,5104,10009596016,7953849472,17963445488,4480,7953859056,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +10009602768,4240,10009607008,7953844464,17963451472,1088,7953849792,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +10009611632,3383936,10012995568,7950458784,17963454352,1632,7953844352,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +10013000288,4400,10013004688,7950454112,17963458800,1472,7950459984,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +10013010256,4128,10013014384,7950448512,17963462896,1280,7950453920,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +10013081440,3632,10013085072,7950381920,17963466992,25376,7950410928,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +10013114832,4174096,10017288928,7946204720,17963493648,3513056,7953891872,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +10017309120,3376,10017312496,7949695520,17967008016,2592,7949701488,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +10017316448,2480320,10019796768,7947215312,17967012080,1984,7949697616,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +10019813408,4352,10019817760,7947198416,17967016176,48640,7947251408,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +10019819968,24896064,10044716032,7922351344,17967067376,3444384,7950691792,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +10044716608,12393200,10057109808,7913404480,17970514288,3840,7925801520,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +10057110592,6272,10057116864,7913403472,17970520336,2452608,7915862352,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +10057165088,2312480,10059477568,7913497296,17972974864,1856,7915811632,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +10059523488,3740880,10063264368,7909714560,17972978928,25941344,7939396784,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +10063367760,3646448,10067014208,7931908784,17998922992,11825184,7947380416,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +10072089968,5698240,10077788208,7932963072,18010751280,2374368,7941035680,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +10077802720,242529136,10320331856,7692796064,18013127920,3870592,7939195792,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +10320338112,2163248,10322501360,7694499360,18017000720,3897536,7700560144,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +10322510048,3520,10322513568,7698386512,18020900080,5456384,7703846416,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +10322521440,2912,10322524352,7703833648,18026358000,5682784,7709519344,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +10322536160,3152,10322539312,7709503904,18032043216,243159328,7952666384,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +10322586624,6128,10322592752,7952612768,18275205520,2185664,7954804560,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +10322593184,2725120,10325318304,7952075312,18277393616,1984,7954802416,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +10325327632,4896,10325332528,7952065216,18277397744,2976,7952073088,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +10325341008,7976736,10333317744,7944084256,18277402000,1536,7952062528,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +10333346592,5436928,10338783520,7938622416,18277405936,50944,7944110288,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +10338800832,3461888,10342262720,7935195440,18277458160,2771296,7941428624,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +10342294256,4528,10342298784,7937933360,18280232144,1856,7937939744,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +10342343488,4128,10342347616,7937888656,18280236272,7895424,7945788208,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +10342387984,3056,10342391040,7945742352,18288133392,5527488,7951272896,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +10342403744,3366192,10345769936,7947892992,18293662928,3679488,7954938672,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +10345789488,3632,10345793120,7951552144,18297345264,2752,7951558528,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +10345797184,2402448,10348199632,7949149696,18297349328,2240,7951554384,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +10348216320,4144,10348220464,7949132960,18297353424,48032,7949185136,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +10348222432,34547792,10382770224,7914633760,18297403984,4093952,7953275504,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +10382770800,142192,10382912992,7918587696,18301500688,1920,7918731808,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +10382913600,12656032,10395569632,7905935152,18301504784,2438080,7921029264,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +10395606256,5072,10395611328,7908333616,18303944944,1888,7908340576,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +10395651008,10406832,10406057840,7897891200,18303949040,34779168,7943077200,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +10406089488,4448,10406093936,7932637344,18338731280,141376,7932783168,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +10406095440,4544,10406099984,7932775712,18338875696,11911488,7944691744,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +10406100768,15667616,10421768384,7929020432,18350788816,2272,7944690320,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +10421769648,5466464,10427236112,7923556832,18350792944,10935616,7939958912,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +10427250736,3472,10427254208,7934477104,18361731312,7584,7934488160,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +10427260624,3056,10427263680,7934491280,18361754960,8672,7934503008,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +10427291888,4336,10427296224,7934470000,18361766224,16098720,7950573056,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +10427326640,4480,10427331120,7950536384,18377867504,5481952,7956022816,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +10427349184,4992,10427354176,7955997936,18383352112,1952,7956004880,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +10427360864,6576,10427367440,7955988704,18383356144,1440,7955996720,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +10427371968,4992,10427376960,7955983344,18383360304,1312,7955989648,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +10427382512,5408,10427387920,7955976416,18383364336,3648,7955985472,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +10427392432,5312,10427397744,7955972736,18383370480,1312,7955979360,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +10427411216,7392,10427418608,7955955936,18383374544,4896,7955968224,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +10427426336,6000,10427432336,7955948384,18383380720,1120,7955955504,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +10427437232,3383152,10430820384,7952563056,18383383440,1632,7955947840,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +10430825344,4096,10430829440,7952558448,18383387888,1472,7952564016,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +10430834912,3536,10430838448,7952553632,18383392080,1472,7952558640,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +10430906208,4352,10430910560,7952485616,18383396176,26816,7952516784,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +10430937888,3461120,10434399008,7949025744,18383424752,3515712,7956002576,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +10434421152,3232,10434424384,7952518928,18386943312,2400,7952524560,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +10434428384,2715872,10437144256,7949803120,18386947376,2048,7952521040,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +10437161424,3840,10437165264,7949786144,18386951408,49056,7949839040,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +10437167840,25295920,10462463760,7924538848,18387002608,3425184,7953259952,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +10462464224,12040320,10474504544,7915925488,18390430032,3808,7927969616,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +10474505904,3648,10474509552,7915926496,18390436048,2433728,7918363872,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +10474558016,2404080,10476962096,7915909184,18392871280,1856,7918315120,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +10477007616,3774624,10480782240,7912093008,18392875248,25886688,7941754320,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +10480880432,3807488,10484687920,7934077120,18418765040,11822752,7949707360,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +10489782672,5707328,10495490000,7935100416,18430590416,2358144,7943165888,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +10495504576,242555424,10738060000,7694891536,18432951536,3870208,7941317168,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +10738064944,2168880,10740233824,7696590480,18436824304,3901312,7702660672,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +10740242016,3424,10740245440,7700482448,18440727888,5449536,7705935408,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +10740253232,3200,10740256432,7705922368,18446178800,5678048,7711603616,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +10740267424,3536,10740270960,7711588736,18451859696,244356320,7955948592,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +10740309120,7584,10740316704,7955902224,18696218928,2190848,7958100656,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +10740317328,2733648,10743050976,7955360240,18698411216,1952,7958095840,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +10743059760,5200,10743064960,7955350480,18698415440,2944,7955358624,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +10743071280,7926752,10750998032,7947423456,18698421488,1536,7955351744,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +10751021888,5496224,10756518112,7941907472,18698425584,49888,7947453584,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +10756533936,3806368,10760340304,7938137600,18698477904,2770112,7944714080,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +10760370896,4336,10760375232,7940874640,18701249872,1888,7940880864,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +10760416384,4608,10760420992,7940832624,18701253616,8399808,7949237040,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +10760459952,3376,10760463328,7949191472,18709654800,5519104,7954713952,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +10760475296,3736560,10764211856,7950964448,18715176304,3799840,7958500848,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +10764230432,3392,10764233824,7954744464,18718978288,4928,7954752784,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +10764237888,2407200,10766645088,7952339504,18718984592,2112,7954748816,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +10766661584,3248,10766664832,7952323696,18718988528,46272,7952373216,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +10766666720,34290256,10800956976,7918079648,18719036624,3862208,7956232112,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +10800957488,141824,10801099312,7921801920,18722901232,1952,7921945696,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +10801100576,12269712,10813370288,7909535040,18722905328,2437024,7924241776,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +10813405408,3552,10813408960,7911934704,18725343664,1888,7911940144,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +10813449504,10705472,10824154976,7901192624,18725347600,34519392,7946417488,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +10824185456,4800,10824190256,7935679488,18759869744,140416,7935824704,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +10824191840,4736,10824196576,7935816464,18760013040,11940640,7947761840,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +10824196976,15747040,10839944016,7932010912,18771954928,2400,7947760352,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +10839944912,5468144,10845413056,7926546064,18771959120,11079968,7943094176,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +10845427776,2992,10845430768,7937609984,18783040752,2496,7937615472,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +10845436992,3232,10845440224,7937604624,18783044848,2144,7937610000,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +10845463472,3664,10845467136,7937581808,18783048944,15949184,7953534656,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +10845493456,5712,10845499168,7953500624,18798999792,5503712,7959010048,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +10845518640,4272,10845522912,7958981904,18804504816,1920,7958988096,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +10845528528,3024,10845531552,7958977488,18804509040,1440,7958981952,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +10845535536,4240,10845539776,7958973328,18804513104,1312,7958978880,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +10845544752,4768,10845549520,7958967584,18804517104,3456,7958975808,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +10845553792,4704,10845558496,7958964848,18804523344,1440,7958970992,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +10845570800,8176,10845578976,7958948368,18804527344,4800,7958961344,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +10845584320,5568,10845589888,7958943824,18804533712,1088,7958950480,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +10845594880,3400304,10848995184,7955542496,18804537680,1632,7958944432,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +10848999600,3200,10849002800,7955538848,18804541648,1504,7955543552,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +10849008000,3136,10849011136,7955533424,18804544560,1280,7955537840,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +10849078496,3600,10849082096,7955466080,18804548176,25920,7955495600,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +10849106608,3388640,10852495248,7952081312,18804576560,3549088,7959019040,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +10852515584,3328,10852518912,7955609840,18808128752,2400,7955615568,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +10852522816,2655088,10855177904,7952955008,18808132912,2176,7955612272,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +10855194528,2912,10855197440,7952939504,18808136944,48352,7952990768,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +10855199632,25476928,10880676560,7927510560,18808187120,3443040,7956430528,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +10880677024,11809264,10892486288,7919145536,18811631824,3840,7930958640,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +10892487120,3728,10892490848,7919147152,18811638000,2447008,7921597888,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +10892535344,2348800,10894884144,7919203264,18814087408,1920,7921553984,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +10894930432,3903680,10898834112,7915257488,18814091600,25021376,7944182544,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +10898926080,3822512,10902748592,7936367552,18839116144,11827520,7952017584,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +10907988240,5662752,10913650992,7937296576,18850947568,2414464,7945373792,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +10913665552,242102864,11155768416,7697596656,18853365072,3933056,7943632576,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +11155775632,2160320,11157935952,7699365248,18857301200,3838272,7705363840,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +11157944656,3344,11157948000,7703193200,18861141200,5452544,7708649088,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +11157955712,2976,11157958688,7708636368,18866595056,5752064,7714391408,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +11157970944,3360,11157974304,7714374640,18872348944,242752640,7957130640,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +11158029168,8016,11158037184,7957067376,19115104560,2175072,7959250464,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +11158037792,2721248,11160759040,7956522480,19117281520,1952,7959245680,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +11160772176,5584,11160777760,7956507952,19117285712,2976,7956516512,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +11160785152,7870928,11168656080,7948635648,19117291728,1632,7956508208,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +11168688352,5474944,11174163296,7943132528,19117295824,49280,7948656752,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +11174181136,3657488,11177838624,7939508400,19117347024,2774432,7945940320,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +11177872496,4000,11177876496,7942246624,19120123120,1888,7942252512,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +11177920672,4176,11177924848,7942202368,19120127216,7918048,7950124592,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +11177967232,2656,11177969888,7950078000,19128047888,5545440,7955626096,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +11177981296,3964352,11181945648,7951650240,19133595888,3704416,7959319008,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +11181967776,3856,11181971632,7955330112,19137301744,6656,7955340624,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +11181975712,2399984,11184375696,7952934240,19137309936,5088,7955339312,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +11184393456,4064,11184397520,7952920576,19137318096,75392,7953000032,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +11184399440,34796224,11219195664,7918199264,19137394928,3973760,7956969248,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +11219196320,145248,11219341568,7922029552,19141371120,1952,7922176752,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +11219343280,11913056,11231256336,7910118880,19141375216,2437984,7924469920,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +11231293280,4416,11231297696,7912516880,19143814576,1888,7912523184,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +11231338640,10712864,11242051504,7901766976,19143818480,34589248,7947069088,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +11242089488,4624,11242094112,7936315120,19178409232,140128,7936459872,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +11242096160,6704,11242102864,7936448672,19178551536,11918368,7948373744,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +11242103264,16270064,11258373328,7932098560,19190471888,2176,7948370800,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +11258374400,5468704,11263843104,7926632880,19190475984,10877600,7942979184,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +11263859136,3056,11263862192,7937492896,19201355088,2048,7937498000,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +11263869040,3264,11263872304,7937486784,19201359088,864,7937490912,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +11263901008,4784,11263905792,7937455440,19201361232,16805760,7954265984,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +11263933664,8880,11263942544,7954226528,19218169072,5480000,7959715408,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +11263959952,4384,11263964336,7959686208,19223650544,1952,7959692544,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +11263973296,5600,11263978896,7959675744,19223654640,1408,7959682752,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +11263984384,5648,11263990032,7959668704,19223658736,3744,7959678096,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +11263995600,6832,11264002432,7959662416,19223664848,3456,7959672704,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +11264006688,3792,11264010480,7959660576,19223671056,1344,7959665712,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +11264026688,6864,11264033552,7959641568,19223675120,4448,7959652880,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +11264040016,6224,11264046240,7959635024,19223681264,1088,7959642336,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +11264050960,3376224,11267427184,7956258144,19223685328,1664,7959636032,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +11267432240,3552,11267435792,7956253664,19223689456,1472,7956258688,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +11267441520,3376,11267444896,7956248656,19223693552,1280,7956253312,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +11267518640,3456,11267522096,7956174464,19223696560,27104,7956205024,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +11267548304,3379696,11270928000,7952798320,19223726320,3516032,7959694048,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +11270947936,3376,11270951312,7956293472,19227244784,2400,7956299248,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +11270955440,2416896,11273372336,7953876544,19227248880,2016,7956295456,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +11273389664,4256,11273393920,7953859088,19227253008,48160,7953911504,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +11273396192,25907280,11299303472,7927999776,19227303248,3527936,7957434992,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +11299304000,11798912,11311102912,7919730992,19230833904,3616,7931533520,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +11311103744,5120,11311108864,7919731184,19230840048,2491008,7922227312,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +11311153232,2338816,11313492048,7919840416,19233332464,1856,7922181088,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +11313539664,3894880,11317434544,7915902016,19233336560,26017728,7945814624,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +11317517952,3871856,11321389808,7937966592,19259356400,11800000,7953638448,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +11326581024,5674672,11332255696,7938904576,19271160272,2401824,7946981072,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +11332270880,242682720,11574953600,7698609776,19273563376,3883328,7945175824,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +11574957984,2170528,11577128512,7700319920,19277448432,3884192,7706374640,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +11577137088,3376,11577140464,7704194464,19281334928,5423936,7709621776,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +11577148000,2960,11577150960,7709609728,19286760688,5687360,7715300048,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +11577163440,2768,11577166208,7715284848,19292451056,241628288,7956915904,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +11577206880,5952,11577212832,7956869488,19534082320,2169824,7959045264,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +11577213936,2729776,11579943712,7956310480,19536254192,1952,7959042208,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +11579952896,5344,11579958240,7956300048,19536258288,2944,7956308336,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +11579964608,7972592,11587937200,7948325312,19536262512,1536,7956299440,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +11587962352,5457488,11593419840,7942846640,19536266480,46432,7948350560,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +11593435824,3691792,11597127616,7939187984,19536315600,2777056,7945656832,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +11597158368,4160,11597162528,7941933264,19539095792,1920,7941939344,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +11597202656,3680,11597206336,7941893552,19539099888,7852032,7949749264,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +11597245680,2880,11597248560,7949706432,19546954992,5480384,7955189696,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +11597260208,3930336,11601190544,7951247968,19552438512,3482688,7958660992,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +11601209328,3520,11601212848,7954711456,19555924304,2336,7954717312,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +11601216976,2402512,11603619488,7952309008,19555928496,2048,7954713568,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +11603636736,3792,11603640528,7952291968,19555932496,47680,7952343440,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +11603642784,34326752,11637969536,7918012016,19555981552,3432384,7955771152,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +11637970176,143776,11638113952,7921302096,19559416048,1952,7921447824,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +11638115056,11933456,11650048512,7909371696,19559420208,2433024,7923738176,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +11650082816,4144,11650086960,7911768352,19561855312,1888,7911774384,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +11650129216,10698336,11660827552,7901031888,19561859440,34484864,7946215088,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +11660863120,3504,11660866624,7935478992,19596345616,140064,7935622560,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +11660868192,3600,11660871792,7935616128,19596487920,12103776,7947723504,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +11660872400,15999776,11676872176,7931721536,19608593712,2048,7947723360,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +11676872672,5466016,11682338688,7926259056,19608597744,10852384,7942577456,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +11682352992,3248,11682356240,7937095904,19619452144,2048,7937101200,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +11682362192,3056,11682365248,7937090992,19619456240,832,7937094880,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +11682388928,4496,11682393424,7937065056,19619458480,15941600,7953011152,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +11682419952,6288,11682426240,7952976752,19635402992,5533120,7958516160,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +11682443648,5520,11682449168,7958488544,19640937712,2560,7958496624,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +11682455824,5056,11682460880,7958480928,19640941808,1440,7958487424,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +11682465232,3248,11682468480,7958477680,19640946160,1312,7958482240,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +11682472800,4880,11682477680,7958472352,19640950032,9408,7958486640,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +11682482416,4144,11682486560,7958475760,19640962320,1408,7958481312,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +11682498624,3600,11682502224,7958464160,19640966384,4544,7958472304,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +11682508304,5520,11682513824,7958458800,19640972624,1088,7958465408,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +11682518144,3398384,11685916528,7955060096,19640976624,11488,7958469968,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +11685921136,3328,11685924464,7955066464,19640990928,10656,7955080448,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +11685929936,3680,11685933616,7955069632,19641003248,1760,7955075072,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +11686002608,3376,11686005984,7955001488,19641007472,34688,7955039552,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +11686030896,3386544,11689417440,7951627888,19641045328,3517632,7958532064,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +11689438448,3376,11689441824,7955122896,19644564720,2368,7955128640,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +11689445776,2414896,11691860672,7952708240,19644568912,2080,7955125216,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +11691877680,4256,11691881936,7952690976,19644572912,50432,7952745664,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +11691884112,26528176,11718412288,7926212848,19644625136,3448832,7956189856,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +11718413040,11804784,11730217824,7917858192,19648076016,3360,7929666336,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +11730218672,4704,11730223376,7917858784,19648082160,2457216,7920320704,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +11730265552,2299376,11732564928,7917977008,19650541936,1888,7920278272,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +11732612032,3838608,11736450640,7914095264,19650545904,25767264,7943701136,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +11736546400,3792736,11740339136,7935976752,19676315888,11886144,7951655632,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +11745724080,5676432,11751400512,7936804080,19688204592,2350784,7944831296,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +11751415552,241260960,11992676512,7697880176,19690556688,3769696,7942910832,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +11992681040,2160784,11994841824,7699487344,19694329168,4067488,7705715616,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +11994850016,3008,11994853024,7703546544,19698399568,5439008,7708988560,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +11994860672,2880,11994863552,7708977552,19703841104,5761760,7714742192,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +11994874528,3184,11994877712,7714726432,19709604144,241841472,7956571088,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +11994912496,5472,11994917968,7956529344,19951447312,2185280,7958720096,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +11994918576,2746912,11997665488,7955970176,19953635664,1952,7958719040,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +11997674384,5056,11997679440,7955960224,19953639664,2976,7955968256,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +11997687744,7856496,12005544240,7948099776,19953644016,1536,7955957808,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +12005566864,5437184,12011004048,7942643904,19953647952,47104,7948128192,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12011021168,3452400,12014473568,7939224432,19953698000,2773376,7945450208,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +12014504912,4416,12014509328,7941964768,19956474096,1856,7941971040,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +12014549744,3712,12014553456,7941924736,19956478192,7878016,7949806464,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +12014592256,3232,12014595488,7949763440,19964358928,5479584,7955246256,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +12014607344,3347184,12017954528,7951885936,19969840464,3491872,7958724992,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +12017973440,3424,12017976864,7955358384,19973335248,2624,7955364432,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12017980944,2405552,12020386496,7952952848,19973339344,2048,7955360448,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12020402704,3312,12020406016,7952937424,19973343440,46688,7952987424,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12020407984,34147376,12054555360,7918836240,19973391600,4083584,7957067200,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +12054555920,139792,12054695712,7922781680,19977477392,2336,7922923808,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +12054696320,12335264,12067031584,7910449968,19977481552,2549760,7925334992,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +12067067616,3856,12067071472,7912961792,19980033264,1888,7912967536,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +12067111744,10733504,12077845248,7902192112,19980037360,34247328,7947172944,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +12077877424,3456,12077880880,7936406240,20014287120,146144,7936555840,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12077882448,3552,12077886000,7936548512,20014434512,12350112,7948902176,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +12077886416,15508976,12093395392,7933391664,20026787056,2208,7948902848,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +12093396160,5630592,12099026752,7927764400,20026791152,10440128,7943835120,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +12099041312,3008,12099044320,7938188592,20037232912,1920,7938193520,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +12099050880,3088,12099053968,7938183008,20037236976,3008,7938189104,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12099080688,4080,12099084768,7938158352,20037243120,14879520,7953041952,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +12099116992,5808,12099122800,7953002112,20052124912,5635680,7958643600,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +12099140368,4576,12099144944,7958617056,20057762000,1984,7958623616,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +12099151600,5888,12099157488,7958608640,20057766128,1440,7958615968,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +12099164048,5232,12099169280,7958600976,20057770256,1440,7958607648,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +12099175600,6880,12099182480,7958591840,20057774320,3072,7958601792,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +12099187920,3568,12099191488,7958588976,20057780464,1792,7958594336,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +12099205584,7216,12099212800,7958571888,20057784688,4544,7958583648,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +12099220304,4240,12099224544,7958566160,20057790704,7584,7958577984,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +12099229024,3415696,12102644720,7955156224,20057800944,1952,7958573872,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +12102649600,3664,12102653264,7955151744,20057805008,2016,7955157424,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +12102658976,4032,12102663008,7955147312,20057810320,1248,7955152592,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12102731760,3472,12102735232,7955078000,20057813232,25632,7955107104,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +12102763296,3785712,12106549008,7951292896,20057841904,3676192,7958754800,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +12106572432,3456,12106575888,7954945248,20061521136,2400,7954951104,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12106579968,2421376,12109001344,7952523984,20061525328,2240,7954947600,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12109018240,3632,12109021872,7952507616,20061529488,49472,7952560720,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12109024048,24918976,12133943024,7927638528,20061581552,3885408,7956442912,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +12133943968,11816512,12145760480,7919708144,20065468624,3648,7931528304,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +12145761104,4224,12145765328,7919709568,20065474896,2445824,7922159616,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +12145810624,2315536,12148126160,7919797024,20067923184,1888,7922114448,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +12148175264,3673040,12151848304,7916078944,20067927248,25171104,7944923088,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +12151958496,3630304,12155588800,7937512464,20093101264,12445568,7953588336,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +12160681776,5663968,12166345744,7939203744,20105549488,2355424,7947223136,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +12166361456,241810560,12408172016,7699734688,20107906704,3719296,7945264544,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +12408177200,2153840,12410331040,7701296560,20111627600,3800608,7707251008,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +12410339888,3392,12410343280,7705087328,20115430608,5700544,7710791264,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +12410351024,2608,12410353632,7710779760,20121133392,5727296,7716509664,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +12410366496,3360,12410369856,7716493808,20126863664,242556320,7959053488,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +12410411216,6880,12410418096,7959003520,20369421616,2181824,7961192224,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +12410419024,2727760,12413146784,7958458032,20371604816,1952,7961187744,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +12413155968,5920,12413161888,7958446928,20371608816,2976,7958455824,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +12413169136,7904880,12421074016,7950539056,20371613072,1504,7958445440,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +12421097872,5436560,12426534432,7945082544,20371616976,53440,7950572544,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12426550144,3452976,12430003120,7941670208,20371673328,2776864,7947900048,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +12430033024,4512,12430037536,7944413936,20374451472,1888,7944420336,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +12430079024,3456,12430082480,7944373056,20374455536,7782656,7952159168,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +12430125088,2960,12430128048,7952111968,20382240016,5478016,7957592944,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +12430139296,3350496,12433489792,7954230640,20387720432,3498272,7961079408,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +12433513184,3440,12433516624,7957703840,20391220464,2336,7957709616,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12433520752,2400656,12435921408,7955303248,20391224656,2208,7957706112,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12435937824,3424,12435941248,7955287248,20391228496,46560,7955337232,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12435943104,33721472,12469664576,7921613232,20391277808,3830208,7959164912,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +12469665200,141584,12469806784,7925303824,20395110608,2208,7925447616,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +12469807456,12804480,12482611936,7912502800,20395114736,2680512,7927987792,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +12482648128,4336,12482652464,7915144096,20397796560,1888,7915150320,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +12482692832,10304464,12492997296,7904803712,20397801008,33924352,7949032528,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +12493029600,3376,12493032976,7938693920,20431726896,141056,7938838352,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12493034512,3376,12493037888,7938831312,20431869200,11924128,7950758816,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +12493038432,15858064,12508896496,7934899296,20443795792,1952,7950759312,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +12508896992,5503856,12514400848,7929398976,20443799824,10514592,7945417424,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +12514415408,3296,12514418704,7939897600,20454316304,1920,7939902816,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +12514425024,3168,12514428192,7939892176,20454320368,864,7939896208,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12514452112,3264,12514455376,7939867232,20454322608,15693888,7955564384,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +12514481248,5584,12514486832,7955531456,20470018288,5522272,7961059312,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +12514509200,4832,12514514032,7961028736,20475542768,2080,7961035648,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +12514520832,3648,12514524480,7961022384,20475546864,1408,7961027440,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +12514528256,3232,12514531488,7961019472,20475550960,1312,7961024016,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +12514536496,4928,12514541424,7961013632,20475555056,10752,7961029312,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +12514546144,4448,12514550592,7961016752,20475567344,1376,7961022576,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +12514563296,5936,12514569232,7961002208,20475571440,4672,7961012816,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +12514575376,5872,12514581248,7960996400,20475577648,1312,7961003584,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +12514586752,3723968,12518310720,7957270992,20475581712,1632,7960996592,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +12518315392,3584,12518318976,7957266800,20475585776,1600,7957271984,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +12518324480,3152,12518327632,7957262432,20475590064,1312,7957266896,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12518398784,3600,12518402384,7957191584,20475593968,31712,7957226896,73,73,0,cudaLaunchKernel,3024 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +12518427728,3783232,12522210960,7953417792,20475628752,3834528,7961035552,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +12522234160,3424,12522237584,7957228224,20479465808,2688,7957234336,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12522241712,2414944,12524656656,7954814144,20479470800,2144,7957231232,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12524673184,4112,12524677296,7954797600,20479474896,49376,7954851088,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12524679424,24946320,12549625744,7929900384,20479526128,3812864,7958659568,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +12549626384,12419104,12562045488,7921295040,20483340528,3968,7933718112,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +12562046160,3536,12562049696,7921296976,20483346672,2433376,7923733888,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +12562093424,2301408,12564394832,7921387936,20485782768,1888,7923691232,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +12564442960,3677760,12568120720,7917666112,20485786832,25033120,7946376992,73,73,0,cuLaunchKernelEx, 296 168 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +12568221584,3636768,12571858352,7938963264,20510821616,12471040,7955071072,73,73,0,cudaLaunchKernel,529340 1 1, 128 1 1,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" +12577385360,5689472,12583074832,7940221088,20523295920,2368672,7948279232,73,73,0,cudaLaunchKernel, 56 37 1, 32 4 1,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" +12583089648,242614848,12825704496,7699963072,20525667568,3739904,7946317824,73,73,0,cudaLaunchKernel,1184 56 1, 512 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +12825708976,2163952,12827872928,7701536336,20529409264,3745888,7707446176,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +12827881296,3200,12827884496,7705272608,20533157104,5564416,7710840224,73,73,0,cudaLaunchKernel, 591 56 1,1024 1 1,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +12827891888,2816,12827894704,7710828928,20538723632,5812416,7716644160,73,73,0,cudaLaunchKernel, 56 1 128, 256 1 1,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" +12827905632,3408,12827909040,7716628800,20544537840,240774464,7957406672,73,73,0,cudaLaunchKernel, 296 56 1, 32 4 1,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" +12827945040,6240,12827951280,7957363008,20785314288,2176096,7959545344,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" +12827951728,2737424,12830689152,7956804176,20787493328,1920,7959543520,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_warp_kernel(const float *, float *, long, float)" +12830697568,4224,12830701792,7956795504,20787497296,3328,7956803056,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +12830708192,7910528,12838618720,7948884592,20787503312,1664,7956796784,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +12838639392,5443872,12844083264,7943424272,20787507536,47328,7948915472,73,73,0,cudaLaunchKernel,8288 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12844098448,3455904,12847554352,7940003232,20787557584,2772384,7946231520,73,73,0,cudaLaunchKernel,529536 1 1, 128 1 1,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" +12847584192,4192,12847588384,7942743376,20790331760,1856,7942749424,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +12847631904,3968,12847635872,7942699888,20790335760,7745440,7950449296,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +12847674080,3232,12847677312,7950407056,20798084368,5472320,7955882608,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +12847688656,3344160,12851032816,7952525824,20803558640,3491552,7959361536,73,73,0,cudaLaunchKernel,37810 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +12851052560,3424,12851055984,7955996544,20807052528,2336,7956002304,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12851060032,2446080,12853506112,7953550704,20807056816,2112,7955998896,73,73,0,cudaLaunchKernel, 63 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12853522048,4080,12853526128,7953534592,20807060720,47616,7953586288,73,73,0,cudaLaunchKernel,6216 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12853527984,34428288,12887956272,7919153568,20807109840,3424352,7957006208,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" +12887956832,142624,12888099456,7922436720,20810536176,1920,7922581264,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +12888100144,11970816,12900070960,7910469312,20810540272,2456224,7924896352,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" +12900106576,4160,12900110736,7912887136,20812997872,2112,7912893408,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +12900151856,10292608,12910444464,7902557504,20813001968,34352224,7947202336,73,73,0,cuLaunchKernelEx, 296 224 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +12910477072,3488,12910480560,7936876640,20847357200,142208,7937022336,73,73,0,cudaLaunchKernel,16576 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12910482112,3440,12910485552,7937016096,20847501648,12776704,7949796240,73,73,0,cudaLaunchKernel, 256 1 1, 128 1 1,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" +12910485968,15710112,12926196080,7934083936,20860280016,2208,7949796256,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" +12926196896,5472784,12931669680,7928614560,20860284240,10460288,7944547632,73,73,0,cudaLaunchKernel,37888 1 1, 256 1 1,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" +12931683824,3296,12931687120,7939059232,20870746352,1920,7939064448,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" +12931693424,3040,12931696464,7939053984,20870750448,832,7939057856,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" +12931720272,4896,12931725168,7939027424,20870752592,15672768,7954705088,73,73,0,cuLaunchKernelEx, 296 42 1, 384 1 1,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu +12931751360,5600,12931756960,7954670928,20886427888,5476768,7960153296,73,73,0,cuLaunchKernelEx,37810 6 1, 128 1 1,_gate_add_kernel +12931794720,5296,12931800016,7960106272,20891906288,1952,7960113520,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +12931813504,5104,12931818608,7960091776,20891910384,1440,7960098320,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +12931822944,4992,12931827936,7960086544,20891914480,3648,7960095184,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +12931834912,3584,12931838496,7960082096,20891920592,3456,7960089136,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +12931843712,6256,12931849968,7960076800,20891926768,1344,7960084400,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" +12931863456,6016,12931869472,7960061392,20891930864,4512,7960071920,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +12931874880,6160,12931881040,7960055968,20891937008,1120,7960063248,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +12931885456,3419808,12935305264,7956635872,20891941136,1632,7960057312,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" +12935310032,3440,12935313472,7956631760,20891945232,1760,7956636960,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +12935319248,3824,12935323072,7956626224,20891949296,1280,7956631328,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12935393616,3472,12935397088,7956556272,20891953360,4640,7956564384,73,73,0,cudaLaunchKernel, 336 1 1, 2 32 1,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" +12935413344,4165712,12939579056,7952380480,20891959536,3591232,7960137424,73,73,0,cudaLaunchKernel,37296 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +12939595600,4528,12939600128,7955952624,20895552752,1952,7955959104,73,73,0,cudaLaunchKernel, 6 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +12939605488,2413968,12942019456,7953537392,20895556848,5457248,7961408608,73,73,0,cudaLaunchKernel,391608 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12942027328,8912,12942036240,7958979552,20901015792,6822208,7965810672,73,73,0,cudaLaunchKernel,783216 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12942048464,24916112,12966964576,7940876176,20907840752,45056,7965837344,73,73,0,cudaLaunchKernel, 414 1 1, 32 4 1,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" +12966971264,12477936,12979449200,7928438656,20907887856,6688,7940923280,73,73,0,cudaLaunchKernel, 6 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +12979453504,3792,12979457296,7928438880,20907896176,51296,7928493968,73,73,0,cudaLaunchKernel,4347 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12979463552,2335072,12981798624,7926151792,20907950416,61632,7928548496,73,73,0,cudaLaunchKernel,8694 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +12981866240,3651664,12985517904,7922496928,20908014832,4410976,7930559568,73,73,0,cudaLaunchKernel, 2 583 1, 8 16 1,"void magma_sgemmEx_kernel(int, int, int, Tensor, int, Tensor, int, Tensor, int, Tensor, int, int, int, const T1 *, const T1 *, T1, T1, int, cublasLtEpilogue_t, int, const void *, long)" +12985629840,3632144,12989261984,7923166384,20912428368,99968,7926898496,73,73,0,cuLaunchKernel, 13 1 3, 128 1 1,void cutlass::Kernel2(T1::Params) +12989265920,5139376,12994405296,7918125376,20912530672,4000,7923268752,73,73,0,cudaLaunchKernel, 1 26 1, 32 16 1,"void cublasLt::splitKreduce_kernel<(int)32, (int)16, int, float, float, float, float, (bool)0, float, float, float, (bool)1, (bool)1, (bool)0, (bool)0>(cublasLt::cublasSplitKParams, const T4 *, const T10 *, T9 *, T5 *, const T6 *, const T6 *, const T11 *, const T4 *, T11 *, void *, long, T6 *, int *, T6 *, T6 *, const T6 *, const T6 *, const T6 *, const T6 *, const T6 *)" +12994440912,5851392,13000292304,7912244512,20912536816,59776,7918155680,73,73,0,cudaLaunchKernel,3497 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +13000296656,243250816,13243547472,7669051776,20912599248,62272,7912364864,73,73,0,cudaLaunchKernel,6993 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +13243557840,2148320,13245706160,7666956672,20912662832,1280,7669106272,73,73,0,cudaLaunchKernel, 13 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +13245724144,3328,13245727472,7666939392,20912666864,100448,7667043168,73,73,0,cudaLaunchKernel,13986 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +13245743984,7392,13245751376,7667018016,20912769392,86400,7667111808,73,73,0,cudaLaunchKernel,3497 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::neg_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +13245758144,2816,13245760960,7667096464,20912857424,1856,7667101136,73,73,0,cudaLaunchKernel, 13 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +13245765808,5008,13245770816,7667090608,20912861424,1472,7667097088,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +13245781872,2852800,13248634672,7664230944,20912865616,5600,7667089344,73,73,0,cudaLaunchKernel, 26 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +13248646240,7456,13248653696,7664220016,20912873712,1312,7664228784,73,73,0,cudaLaunchKernel, 13 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +13248661136,7898176,13256559312,7656318592,20912877904,1536,7664218304,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" +13256563984,5458160,13262022144,7650859920,20912882064,1216,7656319296,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" +13262026592,3464144,13265490736,7647395264,20912886000,1376,7650860784,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" +13265503696,6336,13265510032,7647380288,20912890320,1664,7647388288,73,73,0,cudaLaunchKernel, 13 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::neg_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 9)]::operator ()() const::[lambda(c10::BFloat16) (instance 1)], std::array>(int, T2, T3)" +13265514416,3440,13265517856,7647376432,20912894288,3360,7647383232,73,73,0,cudaLaunchKernel, 26 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +13265525680,22896,13265548576,7647351760,20912900336,4128,7647378784,73,73,0,cudaLaunchKernel, 26 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl_nocast>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +13265553728,3416288,13268970016,7643936560,20912906576,2272,7647355120,73,73,0,cudaLaunchKernel, 26 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" +13268977488,7328,13268984816,7643925760,20912910576,72864,7644005952,73,73,0,cudaLaunchKernel,13986 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +13268990016,2419488,13271409504,7641576048,20912985552,162432,7644157968,73,73,0,cudaLaunchKernel,3497 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add, std::array>(int, T2, T3)" +13271413136,3360,13271416496,7641732736,20913149232,2240,7641738336,73,73,0,cudaLaunchKernel, 52 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +13271419776,34474656,13305894432,7607258832,20913153264,2016,7641735504,73,73,0,cudaLaunchKernel, 13 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add, std::array>(int, T2, T3)" +13305926784,118912,13306045696,7607111376,20913157072,2848,7607233136,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel::CompareEqFunctor>, std::array, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithoutCast, at::native::memory::StoreWithoutCast>(int, T1, T2, T4, T5, T6, T7)" +20913225648,18000,,,20913239056,159936,173344,73,73,0,cudaLaunchKernel,3497 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add, std::array>(int, T2, T3)" +20913311408,35360,20913346768,53984,20913400752,83872,173216,73,73,0,cudaLaunchKernel,13986 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20913363152,18032,20913381184,105616,20913486800,1184,124832,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithoutCast, at::native::memory::StoreWithoutCast>(int, T1, T2, T4, T5, T6, T7)" +20913400864,6560,20913407424,81584,20913489008,88064,176208,73,73,0,cudaLaunchKernel,13986 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20913431520,11424,20913442944,135248,20913578192,172960,319632,73,73,0,cudaLaunchKernel,3497 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add, std::array>(int, T2, T3)" +20913495840,13200,20913509040,244000,20913753040,1824,259024,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel::CompareEqFunctor>, std::array, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithoutCast, at::native::memory::StoreWithoutCast>(int, T1, T2, T4, T5, T6, T7)" +20913773120,2928,,,20913773072,2336,2928,73,73,0,cudaLaunchKernel, 13 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add, std::array>(int, T2, T3)" +20913780720,3216,,,20913782096,3904,5280,73,73,0,cudaLaunchKernel, 52 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20913787184,3296,20913790480,4544,20913795024,4928,12768,73,73,0,cudaLaunchKernel, 1 1 1, 128 1 1,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithoutCast, at::native::memory::StoreWithoutCast>(int, T1, T2, T4, T5, T6, T7)" +20913793584,2976,20913796560,4544,20913801104,27520,35040,73,73,0,cudaLaunchKernel, 52 1 1, 128 1 1,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" +20913800032,2736,20913802768,28128,20913830896,1600,32464,73,73,0,cudaLaunchKernel, 13 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add, std::array>(int, T2, T3)" +20913857440,4672,,,20913858544,1120,4672,73,73,0,cudaLaunchKernel, 13 1 1, 128 1 1,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" diff --git a/benchmarks/gb10-post-fc2-warmed-step-stats_nvtx_gpu_proj_trace.csv b/benchmarks/gb10-post-fc2-warmed-step-stats_nvtx_gpu_proj_trace.csv new file mode 100644 index 0000000..4c0cd04 --- /dev/null +++ b/benchmarks/gb10-post-fc2-warmed-step-stats_nvtx_gpu_proj_trace.csv @@ -0,0 +1,2 @@ +Name,Projected Start (ns),Projected Duration (ns),Orig Start (ns),Orig Duration (ns),Style,PID,TID,NumGPUOps,Lvl,NumChild,RangeId,ParentId,RangeStack +:complete_sampling_run,6687024,20907172640,5969088,20907943008,PushPop,73,73,2803,0,0,1,,:1 diff --git a/research/artifact_manifest.json b/research/artifact_manifest.json index 044ef26..156eadf 100644 --- a/research/artifact_manifest.json +++ b/research/artifact_manifest.json @@ -1,37 +1,37 @@ { "metadata": { "algorithm": "SHA-256", - "generated_date": "2026-08-25", - "scope_note": "Local selected scope is the established 167-file set plus 14 retained FC2 scheduling and integration artifacts. Spark records are the complete reproducible current top-level benchmark output set; the audit-retained 279-file/665950155-byte aggregate cannot be reconstructed because its path list was not retained.", + "generated_date": "2026-08-26", + "scope_note": "Local selected scope is the established 167-file set plus 14 retained FC2 scheduling and integration artifacts and 19 post-FC2 production-profile artifacts. Spark records are the complete reproducible current top-level benchmark output set; the audit-retained 279-file/665950155-byte aggregate cannot be reconstructed because its path list was not retained.", "summary": { "local": { - "record_count": 181, - "size_bytes": 230398680, - "expected_record_count": 181, - "expected_size_bytes": 230398680, + "record_count": 200, + "size_bytes": 257150989, + "expected_record_count": 200, + "expected_size_bytes": 257150989, "reconciled": true }, "spark": { - "record_count": 294, - "size_bytes": 567718575, + "record_count": 313, + "size_bytes": 594470884, "expected_record_count": 279, "expected_size_bytes": 665950155, "reconciled": false, - "record_count_delta": 15, - "size_bytes_delta": -98231580 + "record_count_delta": 34, + "size_bytes_delta": -71479271 }, "total": { - "record_count": 475, - "size_bytes": 798117255 + "record_count": 513, + "size_bytes": 851621873 } }, "json_reconciliation": { - "identical": 118, + "identical": 126, "mismatches": 0, "local_only": 40, "spark_only": 107, - "local_total": 158, - "spark_total": 225 + "local_total": 166, + "spark_total": 233 } }, "artifacts": [ @@ -3834,6 +3834,44 @@ "sha256": "9b711de3301244891d04465249db49a9233951928607a50a6ef9823a5f0fab58", "artifact_class": "run_log", "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/warmup-direct-sage3-fp16-1s.log" - } + }, + {"scope": "local", "path": "benchmarks/gb10-post-fc2-block24-profile-20260826.json", "size_bytes": 40955, "sha256": "631da0f3780250e2cd34dd39c5a0dbecc0624c58fefa38048c961fb161f80c86", "artifact_class": "benchmark_json", "location": "S:\\PycharmProjects\\AuthorCompanion\\tmp\\h3-blackwell-runtime\\benchmarks\\gb10-post-fc2-block24-profile-20260826.json"}, + {"scope": "local", "path": "benchmarks/gb10-post-fc2-block24-targeted-20260826.csv", "size_bytes": 50742, "sha256": "6fa41faf2de5697106a9a90520c2f812ffeb900051ae4fdadd4a5ef36dc9fe5f", "artifact_class": "nsight_csv_export", "location": "S:\\PycharmProjects\\AuthorCompanion\\tmp\\h3-blackwell-runtime\\benchmarks\\gb10-post-fc2-block24-targeted-20260826.csv"}, + {"scope": "local", "path": "benchmarks/gb10-post-fc2-block24-targeted-20260826.ncu-rep", "size_bytes": 8696909, "sha256": "f07a5abcd275951096ae5c95522fdbe5670081711bec73a4e381fcee3066af0d", "artifact_class": "nsight_compute_report", "location": "S:\\PycharmProjects\\AuthorCompanion\\tmp\\h3-blackwell-runtime\\benchmarks\\gb10-post-fc2-block24-targeted-20260826.ncu-rep"}, + {"scope": "local", "path": "benchmarks/gb10-post-fc2-block24-targeted-capture-20260826.json", "size_bytes": 487, "sha256": "cb97c3f3c80bf5bee185df51b5c4b0b82bf90a55bb026431b25b2faa7bdd5b15", "artifact_class": "benchmark_json", "location": "S:\\PycharmProjects\\AuthorCompanion\\tmp\\h3-blackwell-runtime\\benchmarks\\gb10-post-fc2-block24-targeted-capture-20260826.json"}, + {"scope": "local", "path": "benchmarks/gb10-post-fc2-block24-targeted-summary-20260826.json", "size_bytes": 7359, "sha256": "2d046da14d90bb97317da176e75491f887732afccd48e9703c9904dd0e0e57f2", "artifact_class": "benchmark_json", "location": "S:\\PycharmProjects\\AuthorCompanion\\tmp\\h3-blackwell-runtime\\benchmarks\\gb10-post-fc2-block24-targeted-summary-20260826.json"}, + {"scope": "local", "path": "benchmarks/gb10-post-fc2-block24-targeted-traffic-20260826.csv", "size_bytes": 21877, "sha256": "4d7bca04ed986897989e9bee5070461d9259b2797779ac74504603fa9a7c5005", "artifact_class": "nsight_csv_export", "location": "S:\\PycharmProjects\\AuthorCompanion\\tmp\\h3-blackwell-runtime\\benchmarks\\gb10-post-fc2-block24-targeted-traffic-20260826.csv"}, + {"scope": "local", "path": "benchmarks/gb10-post-fc2-block24-targeted-traffic-20260826.ncu-rep", "size_bytes": 8082484, "sha256": "27cd225e5ed94977d9437410b42d5a2228e74f01100a737091b0b637b07c42d5", "artifact_class": "nsight_compute_report", "location": "S:\\PycharmProjects\\AuthorCompanion\\tmp\\h3-blackwell-runtime\\benchmarks\\gb10-post-fc2-block24-targeted-traffic-20260826.ncu-rep"}, + {"scope": "local", "path": "benchmarks/gb10-post-fc2-block24-targeted-traffic-capture-20260826.json", "size_bytes": 487, "sha256": "cb97c3f3c80bf5bee185df51b5c4b0b82bf90a55bb026431b25b2faa7bdd5b15", "artifact_class": "benchmark_json", "location": "S:\\PycharmProjects\\AuthorCompanion\\tmp\\h3-blackwell-runtime\\benchmarks\\gb10-post-fc2-block24-targeted-traffic-capture-20260826.json"}, + {"scope": "local", "path": "benchmarks/gb10-post-fc2-production-profile-summary-20260826.json", "size_bytes": 20432, "sha256": "559f9a0ebbe9507fece3612949329632761e3f7e10b42ceb5c87854583b6f0af", "artifact_class": "benchmark_json", "location": "S:\\PycharmProjects\\AuthorCompanion\\tmp\\h3-blackwell-runtime\\benchmarks\\gb10-post-fc2-production-profile-summary-20260826.json"}, + {"scope": "local", "path": "benchmarks/gb10-post-fc2-resident-baseline-20260826.json", "size_bytes": 28079, "sha256": "0b73bed25e8e209101a63b5dde633b7eea38783e251b01dbf2e6cd59bf4ee679", "artifact_class": "benchmark_json", "location": "S:\\PycharmProjects\\AuthorCompanion\\tmp\\h3-blackwell-runtime\\benchmarks\\gb10-post-fc2-resident-baseline-20260826.json"}, + {"scope": "local", "path": "benchmarks/gb10-post-fc2-warmed-step-20260826.nsys-rep", "size_bytes": 856665, "sha256": "286b925c48e7bc2e89d2b68a5c0a07c07c1905419ca7b3f511cc5d7e289f3cbb", "artifact_class": "nsight_systems_report", "location": "S:\\PycharmProjects\\AuthorCompanion\\tmp\\h3-blackwell-runtime\\benchmarks\\gb10-post-fc2-warmed-step-20260826.nsys-rep"}, + {"scope": "local", "path": "benchmarks/gb10-post-fc2-warmed-step-20260826.sqlite", "size_bytes": 7106560, "sha256": "b8f64ac3938b190ecb1ef8ecedfe37278dd779fa8196a5e7220fe0cb90e217fb", "artifact_class": "nsight_sqlite_database", "location": "S:\\PycharmProjects\\AuthorCompanion\\tmp\\h3-blackwell-runtime\\benchmarks\\gb10-post-fc2-warmed-step-20260826.sqlite"}, + {"scope": "local", "path": "benchmarks/gb10-post-fc2-warmed-step-capture-20260826.json", "size_bytes": 1963, "sha256": "402f2f4b4aa0f987f5cf133cc3c09f54bfcee0769c37a4a9981cb0e318fbe71f", "artifact_class": "benchmark_json", "location": "S:\\PycharmProjects\\AuthorCompanion\\tmp\\h3-blackwell-runtime\\benchmarks\\gb10-post-fc2-warmed-step-capture-20260826.json"}, + {"scope": "local", "path": "benchmarks/gb10-post-fc2-warmed-step-nsys-summary-20260826.json", "size_bytes": 2094, "sha256": "8bb61cb70ade2d5f794a21c7cc3ccf58d2bc8d85e3d8abe596437eb9749aafdd", "artifact_class": "benchmark_json", "location": "S:\\PycharmProjects\\AuthorCompanion\\tmp\\h3-blackwell-runtime\\benchmarks\\gb10-post-fc2-warmed-step-nsys-summary-20260826.json"}, + {"scope": "local", "path": "benchmarks/gb10-post-fc2-warmed-step-stats_cuda_api_sum.csv", "size_bytes": 788, "sha256": "e0be4be5700ee6cff7bedf0bc7e9168a2b41a0825034ec8996b7e98e4da8ac8d", "artifact_class": "nsight_csv_export", "location": "S:\\PycharmProjects\\AuthorCompanion\\tmp\\h3-blackwell-runtime\\benchmarks\\gb10-post-fc2-warmed-step-stats_cuda_api_sum.csv"}, + {"scope": "local", "path": "benchmarks/gb10-post-fc2-warmed-step-stats_cuda_gpu_kern_sum.csv", "size_bytes": 18564, "sha256": "f7459b6e3cf004f89b2c42df4e0bf752b25fcbd7b550962e4ea95673d9edbcf5", "artifact_class": "nsight_csv_export", "location": "S:\\PycharmProjects\\AuthorCompanion\\tmp\\h3-blackwell-runtime\\benchmarks\\gb10-post-fc2-warmed-step-stats_cuda_gpu_kern_sum.csv"}, + {"scope": "local", "path": "benchmarks/gb10-post-fc2-warmed-step-stats_cuda_gpu_trace.csv", "size_bytes": 853416, "sha256": "d7eafe6a3da20b5afa49bcb7a531c04c396f47f3ba76934e978857440ab2bd61", "artifact_class": "nsight_csv_export", "location": "S:\\PycharmProjects\\AuthorCompanion\\tmp\\h3-blackwell-runtime\\benchmarks\\gb10-post-fc2-warmed-step-stats_cuda_gpu_trace.csv"}, + {"scope": "local", "path": "benchmarks/gb10-post-fc2-warmed-step-stats_cuda_kern_exec_trace.csv", "size_bytes": 962206, "sha256": "509c9ad616f0834759871f46e302137d979be21a540b6656173f876589c08273", "artifact_class": "nsight_csv_export", "location": "S:\\PycharmProjects\\AuthorCompanion\\tmp\\h3-blackwell-runtime\\benchmarks\\gb10-post-fc2-warmed-step-stats_cuda_kern_exec_trace.csv"}, + {"scope": "local", "path": "benchmarks/gb10-post-fc2-warmed-step-stats_nvtx_gpu_proj_trace.csv", "size_bytes": 242, "sha256": "12bfbbbbe10467badbf6b4a6ca712241f0375223d2798ab0f3ea68051a2c7f8c", "artifact_class": "nsight_csv_export", "location": "S:\\PycharmProjects\\AuthorCompanion\\tmp\\h3-blackwell-runtime\\benchmarks\\gb10-post-fc2-warmed-step-stats_nvtx_gpu_proj_trace.csv"}, + {"scope": "spark", "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-block24-profile-20260826.json", "size_bytes": 40955, "sha256": "631da0f3780250e2cd34dd39c5a0dbecc0624c58fefa38048c961fb161f80c86", "artifact_class": "benchmark_json", "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-block24-profile-20260826.json"}, + {"scope": "spark", "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-block24-targeted-20260826.csv", "size_bytes": 50742, "sha256": "6fa41faf2de5697106a9a90520c2f812ffeb900051ae4fdadd4a5ef36dc9fe5f", "artifact_class": "nsight_csv_export", "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-block24-targeted-20260826.csv"}, + {"scope": "spark", "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-block24-targeted-20260826.ncu-rep", "size_bytes": 8696909, "sha256": "f07a5abcd275951096ae5c95522fdbe5670081711bec73a4e381fcee3066af0d", "artifact_class": "nsight_compute_report", "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-block24-targeted-20260826.ncu-rep"}, + {"scope": "spark", "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-block24-targeted-capture-20260826.json", "size_bytes": 487, "sha256": "cb97c3f3c80bf5bee185df51b5c4b0b82bf90a55bb026431b25b2faa7bdd5b15", "artifact_class": "benchmark_json", "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-block24-targeted-capture-20260826.json"}, + {"scope": "spark", "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-block24-targeted-summary-20260826.json", "size_bytes": 7359, "sha256": "2d046da14d90bb97317da176e75491f887732afccd48e9703c9904dd0e0e57f2", "artifact_class": "benchmark_json", "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-block24-targeted-summary-20260826.json"}, + {"scope": "spark", "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-block24-targeted-traffic-20260826.csv", "size_bytes": 21877, "sha256": "4d7bca04ed986897989e9bee5070461d9259b2797779ac74504603fa9a7c5005", "artifact_class": "nsight_csv_export", "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-block24-targeted-traffic-20260826.csv"}, + {"scope": "spark", "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-block24-targeted-traffic-20260826.ncu-rep", "size_bytes": 8082484, "sha256": "27cd225e5ed94977d9437410b42d5a2228e74f01100a737091b0b637b07c42d5", "artifact_class": "nsight_compute_report", "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-block24-targeted-traffic-20260826.ncu-rep"}, + {"scope": "spark", "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-block24-targeted-traffic-capture-20260826.json", "size_bytes": 487, "sha256": "cb97c3f3c80bf5bee185df51b5c4b0b82bf90a55bb026431b25b2faa7bdd5b15", "artifact_class": "benchmark_json", "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-block24-targeted-traffic-capture-20260826.json"}, + {"scope": "spark", "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-production-profile-summary-20260826.json", "size_bytes": 20432, "sha256": "559f9a0ebbe9507fece3612949329632761e3f7e10b42ceb5c87854583b6f0af", "artifact_class": "benchmark_json", "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-production-profile-summary-20260826.json"}, + {"scope": "spark", "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-resident-baseline-20260826.json", "size_bytes": 28079, "sha256": "0b73bed25e8e209101a63b5dde633b7eea38783e251b01dbf2e6cd59bf4ee679", "artifact_class": "benchmark_json", "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-resident-baseline-20260826.json"}, + {"scope": "spark", "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-warmed-step-20260826.nsys-rep", "size_bytes": 856665, "sha256": "286b925c48e7bc2e89d2b68a5c0a07c07c1905419ca7b3f511cc5d7e289f3cbb", "artifact_class": "nsight_systems_report", "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-warmed-step-20260826.nsys-rep"}, + {"scope": "spark", "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-warmed-step-20260826.sqlite", "size_bytes": 7106560, "sha256": "b8f64ac3938b190ecb1ef8ecedfe37278dd779fa8196a5e7220fe0cb90e217fb", "artifact_class": "nsight_sqlite_database", "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-warmed-step-20260826.sqlite"}, + {"scope": "spark", "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-warmed-step-capture-20260826.json", "size_bytes": 1963, "sha256": "402f2f4b4aa0f987f5cf133cc3c09f54bfcee0769c37a4a9981cb0e318fbe71f", "artifact_class": "benchmark_json", "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-warmed-step-capture-20260826.json"}, + {"scope": "spark", "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-warmed-step-nsys-summary-20260826.json", "size_bytes": 2094, "sha256": "8bb61cb70ade2d5f794a21c7cc3ccf58d2bc8d85e3d8abe596437eb9749aafdd", "artifact_class": "benchmark_json", "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-warmed-step-nsys-summary-20260826.json"}, + {"scope": "spark", "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-warmed-step-stats_cuda_api_sum.csv", "size_bytes": 788, "sha256": "e0be4be5700ee6cff7bedf0bc7e9168a2b41a0825034ec8996b7e98e4da8ac8d", "artifact_class": "nsight_csv_export", "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-warmed-step-stats_cuda_api_sum.csv"}, + {"scope": "spark", "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-warmed-step-stats_cuda_gpu_kern_sum.csv", "size_bytes": 18564, "sha256": "f7459b6e3cf004f89b2c42df4e0bf752b25fcbd7b550962e4ea95673d9edbcf5", "artifact_class": "nsight_csv_export", "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-warmed-step-stats_cuda_gpu_kern_sum.csv"}, + {"scope": "spark", "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-warmed-step-stats_cuda_gpu_trace.csv", "size_bytes": 853416, "sha256": "d7eafe6a3da20b5afa49bcb7a531c04c396f47f3ba76934e978857440ab2bd61", "artifact_class": "nsight_csv_export", "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-warmed-step-stats_cuda_gpu_trace.csv"}, + {"scope": "spark", "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-warmed-step-stats_cuda_kern_exec_trace.csv", "size_bytes": 962206, "sha256": "509c9ad616f0834759871f46e302137d979be21a540b6656173f876589c08273", "artifact_class": "nsight_csv_export", "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-warmed-step-stats_cuda_kern_exec_trace.csv"}, + {"scope": "spark", "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-warmed-step-stats_nvtx_gpu_proj_trace.csv", "size_bytes": 242, "sha256": "12bfbbbbe10467badbf6b4a6ca712241f0375223d2798ab0f3ea68051a2c7f8c", "artifact_class": "nsight_csv_export", "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-warmed-step-stats_nvtx_gpu_proj_trace.csv"} ] } diff --git a/src/h3_blackwell_runtime/runtime.py b/src/h3_blackwell_runtime/runtime.py index 3b0e4e6..1bd60b1 100644 --- a/src/h3_blackwell_runtime/runtime.py +++ b/src/h3_blackwell_runtime/runtime.py @@ -2,6 +2,7 @@ from __future__ import annotations +import hashlib import os import subprocess import time @@ -55,6 +56,11 @@ def _sync() -> None: torch.cuda.synchronize() +def _tensor_sha256(value: torch.Tensor) -> str: + immutable = value.detach().contiguous().view(torch.uint16).cpu() + return hashlib.sha256(immutable.numpy().tobytes()).hexdigest() + + def _ffmpeg_command(loglevel: str, *parts: str) -> list[str]: return ["ffmpeg", "-hide_banner", "-loglevel", loglevel, *parts] @@ -263,9 +269,12 @@ class H3HotRuntime: cache_start_percent: float = 0.0, cache_end_percent: float = 1.0, cache_subsample_factor: int = 2, + benchmark_text_tokens: int | None = None, ) -> dict: stages: list[dict] = [] cache_stats: dict = {} + diagnostics_enabled = benchmark_text_tokens is not None + sampling_step_events: list[dict] | None = [] if diagnostics_enabled else None def timed(stage: str, fn): _sync() @@ -313,7 +322,19 @@ class H3HotRuntime: "frame_count": frame_count, } else: - text = timed("text_conditioned", lambda: self.refiner(self.conditioner(prompt))) + if benchmark_text_tokens is None: + text = timed("text_conditioned", lambda: self.refiner(self.conditioner(prompt))) + else: + text = timed( + "text_conditioned", + lambda: torch.randn( + 1, + benchmark_text_tokens, + 5376, + device=self.config.device, + dtype=torch.bfloat16, + ), + ) pack_kwargs = {} if self.turbo is None: sample = lambda: sample_video_res_multistep( @@ -321,6 +342,7 @@ class H3HotRuntime: return_audio=mux_audio, cache_mode=cache_mode, cache_threshold=cache_threshold, cache_start_percent=cache_start_percent, cache_end_percent=cache_end_percent, cache_subsample_factor=cache_subsample_factor, cache_stats=cache_stats, **pack_kwargs, + sampling_step_events=sampling_step_events, ) else: sample = lambda: sample_video_turbo( @@ -328,11 +350,32 @@ class H3HotRuntime: video_shift=TURBO_VARIANTS[self.turbo]["video_shift"], seed=seed, return_audio=mux_audio, **pack_kwargs, ) + from .fc2_lt import fc2_lt_status + + fc2_before = fc2_lt_status() + if diagnostics_enabled: + torch.cuda.reset_peak_memory_stats() sampled = timed("sampled", sample) + sampling_peak_allocated = torch.cuda.max_memory_allocated() if diagnostics_enabled else None + sampling_peak_reserved = torch.cuda.max_memory_reserved() if diagnostics_enabled else None + fc2_after = fc2_lt_status() + sampling_steps = [ + { + "step": row["step"], + "seconds": row["start"].elapsed_time(row["end"]) / 1000.0, + } + for row in sampling_step_events or () + ] if mux_audio: latent, audio_latent = sampled else: latent, audio_latent = sampled, None + latent_checksums = None + if diagnostics_enabled: + latent_checksums = { + "video_sha256": _tensor_sha256(latent), + "audio_sha256": _tensor_sha256(audio_latent) if audio_latent is not None else None, + } source_width, source_height = width, height if upscale_scale is not None: latent = timed( @@ -411,6 +454,18 @@ class H3HotRuntime: "keep_intermediates": keep_intermediates, "vae_dtype": self.config.vae_dtype, "vae_tile_size": self.config.vae_tile_size, + "benchmark_text_tokens": benchmark_text_tokens, + "sampling_seconds": next( + stage["seconds"] for stage in stages if stage["stage"] == "sampled" + ), + "sampling_steps": sampling_steps, + "sampling_peak_allocated_bytes": sampling_peak_allocated, + "sampling_peak_reserved_bytes": sampling_peak_reserved, + "latent_checksums": latent_checksums, + "fc2_dispatch_delta": { + name: fc2_after[name] - fc2_before[name] + for name in ("attempts", "successes", "fallbacks") + }, "stages": stages, "cache": cache_stats, "request_seconds": sum(stage["seconds"] for stage in stages), diff --git a/src/h3_blackwell_runtime/sampler.py b/src/h3_blackwell_runtime/sampler.py index d33a6ab..14a0b02 100644 --- a/src/h3_blackwell_runtime/sampler.py +++ b/src/h3_blackwell_runtime/sampler.py @@ -116,6 +116,7 @@ def sample_video_res_multistep( cache_stats: dict | None = None, audio_step_trace: list[dict] | None = None, sampling_stage_trace: list[dict] | None = None, + sampling_step_events: list[dict] | None = None, ) -> torch.Tensor: """Direct H3 beta/RES sampling with Comfy-equivalent joint AV carry semantics.""" sigmas = beta_sigmas(steps, device=video.device) @@ -138,6 +139,10 @@ def sample_video_res_multistep( "cumulative_rate": 0.0, } for index, sigma in enumerate(sigmas[:-1], start=1): + step_start_event = None + if sampling_step_events is not None: + step_start_event = torch.cuda.Event(enable_timing=True) + step_start_event.record() step_started = time.perf_counter() stage_row = None if sampling_stage_trace is not None: @@ -256,6 +261,14 @@ def sample_video_res_multistep( }) video_history, audio_history = video_denoised, audio_denoised video_history_sigma = audio_history_sigma = sigma_down + if sampling_step_events is not None: + step_end_event = torch.cuda.Event(enable_timing=True) + step_end_event.record() + sampling_step_events.append({ + "step": index, + "start": step_start_event, + "end": step_end_event, + }) if stage_row is not None: torch.cuda.synchronize(video.device) stage_row["step_total"] = time.perf_counter() - profiled_step_started diff --git a/tests/test_turbo.py b/tests/test_turbo.py index 1a29a71..d3ee020 100644 --- a/tests/test_turbo.py +++ b/tests/test_turbo.py @@ -115,6 +115,43 @@ class TurboLoraContracts(unittest.TestCase): torch.testing.assert_close(trace[0]["audio_after"], torch.full_like(audio, 4.0)) torch.testing.assert_close(final_audio, torch.ones_like(audio)) + def test_base_sampler_records_deferred_step_events_without_synchronizing(self): + video = torch.zeros(1, 1, 1, 1, 1) + audio = torch.zeros(1, 32, 2, 1) + events = [] + + class FakeEvent: + def __init__(self, *, enable_timing): + self.enable_timing = enable_timing + self.recorded = False + + def record(self): + self.recorded = True + + def packer(*args, **kwargs): + return (None, None, None, None, None, None) + + def model(*args): + return torch.ones(1), torch.ones(1) + + with ( + patch("h3_blackwell_runtime.sampler.beta_sigmas", return_value=torch.tensor([1.0, 0.0])), + patch("h3_blackwell_runtime.sampler.unpatchify_video", return_value=torch.zeros_like(video)), + patch("h3_blackwell_runtime.sampler._unpack_audio", return_value=torch.ones_like(audio)), + patch("h3_blackwell_runtime.sampler.torch.cuda.Event", side_effect=FakeEvent), + patch("h3_blackwell_runtime.sampler.torch.cuda.synchronize") as synchronize, + ): + sample_video_res_multistep( + model, packer, torch.empty(0), video, audio, + steps=1, return_audio=True, sampling_step_events=events, + ) + + self.assertEqual(len(events), 1) + self.assertEqual(events[0]["step"], 1) + self.assertTrue(events[0]["start"].recorded) + self.assertTrue(events[0]["end"].recorded) + synchronize.assert_not_called() + if __name__ == "__main__": unittest.main() diff --git a/tools/benchmark_hot_runtime.py b/tools/benchmark_hot_runtime.py new file mode 100644 index 0000000..fbd9037 --- /dev/null +++ b/tools/benchmark_hot_runtime.py @@ -0,0 +1,164 @@ +"""Establish a repeated canonical baseline through the resident H3 API.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import statistics +import time +from pathlib import Path +from urllib.request import Request, urlopen + + +DEFAULT_PROMPT = ( + "A playful orange tabby cat starts in an ordinary cozy living room in a normal house, " + "afternoon light, sofa and rug. The cat crouches, jumps, and does one clean athletic " + "backflip in slow motion. As the backflip completes there is a sharp cinematic cut: " + "the cat lands perfectly on a glowing neon disco dance floor wearing oversized black " + "sunglasses. Mirror ball reflections, colorful lights, joyful party energy, stylish " + "and funny, clear before-and-after transformation." +) +EXPECTED_VIDEO_SHA256 = "c62d23a42972eab907ba42f93c50247ff17a9c454b4a53fe93d2e34f9fefe578" +EXPECTED_AUDIO_SHA256 = "852005383770480a6503504e1ffec86dd1fb63a69c6400f92da18e39e0986de2" + + +def get_json(url: str, timeout: float = 30.0) -> dict: + with urlopen(url, timeout=timeout) as response: + return json.loads(response.read().decode()) + + +def post_json(url: str, payload: dict, timeout: float) -> dict: + request = Request( + url, + data=json.dumps(payload).encode(), + headers={"Content-Type": "application/json"}, + ) + with urlopen(request, timeout=timeout) as response: + return json.loads(response.read().decode()) + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--server", default="http://127.0.0.1:8001") + parser.add_argument("--output", type=Path, required=True) + parser.add_argument("--runs", type=int, default=3) + parser.add_argument("--timeout", type=float, default=1200.0) + parser.add_argument("--source-commit", default="a29b8960b0f887c20e74dafa16a24c37d6508b4e") + parser.add_argument("--image", required=True) + args = parser.parse_args() + if args.runs < 3: + raise ValueError("the authoritative baseline requires at least three measured runs") + + ready_before = get_json(f"{args.server}/ready") + if not ready_before.get("ready"): + raise RuntimeError("resident runtime is not ready") + if ready_before["runtime"]["current_attention"] != "sage2": + raise RuntimeError("resident runtime is not using Sage2") + if not ready_before["runtime"]["fc2_lt"]["enabled"]: + raise RuntimeError("guarded FC2 schedule is disabled") + + def payload(label: str) -> dict: + return { + "prompt": DEFAULT_PROMPT, + "output": f"/output/h3-blackwell-runtime/post-fc2-{label}.mp4", + "width": 1344, + "height": 768, + "frames": 124, + "steps": 12, + "seed": 440420, + "attention": "sage2", + "turbo": None, + "cache_mode": None, + "mux_audio": True, + "keep_intermediates": False, + "benchmark_text_tokens": 100, + } + + warmup = post_json(f"{args.server}/generate", payload("canonical-warmup"), args.timeout) + measured = [] + for index in range(1, args.runs + 1): + started = time.perf_counter() + response = post_json( + f"{args.server}/generate", payload(f"canonical-run-{index}"), args.timeout, + ) + response["client_wall_seconds"] = time.perf_counter() - started + measured.append(response) + + for name, response in [("warmup", warmup), *[ + (f"run_{index}", value) for index, value in enumerate(measured, start=1) + ]]: + dispatch = response["fc2_dispatch_delta"] + if dispatch != {"attempts": 600, "successes": 600, "fallbacks": 0}: + raise RuntimeError(f"{name} FC2 dispatch validation failed: {dispatch}") + if len(response["sampling_steps"]) != 12: + raise RuntimeError(f"{name} did not report 12 sampling steps") + + checksum_pairs = [ + (row["latent_checksums"]["video_sha256"], row["latent_checksums"]["audio_sha256"]) + for row in measured + ] + exact_parity = len(set(checksum_pairs)) == 1 + if not exact_parity: + raise RuntimeError(f"measured latent checksums differ: {checksum_pairs}") + expected_checksums = (EXPECTED_VIDEO_SHA256, EXPECTED_AUDIO_SHA256) + if checksum_pairs[0] != expected_checksums: + raise RuntimeError( + f"latent checksums differ from canonical reference: " + f"expected {expected_checksums}, got {checksum_pairs[0]}" + ) + + sampling_seconds = [row["sampling_seconds"] for row in measured] + report = { + "name": "gb10-post-fc2-resident-baseline", + "source_commit": args.source_commit, + "image": args.image, + "measurement_date": "2026-08-26", + "workload": { + "resolution": [1344, 768], + "frames": 124, + "steps": 12, + "seed": 440420, + "attention": "sage2", + "prompt": DEFAULT_PROMPT, + "prompt_sha256": hashlib.sha256(DEFAULT_PROMPT.encode()).hexdigest(), + }, + "measurement_policy": { + "canonical_warmup_runs": 1, + "measured_runs": args.runs, + "authoritative_timing": "median synchronized resident sampling_seconds", + "step_timing": "deferred CUDA event elapsed time; no per-step synchronization", + "profiling": False, + }, + "sampling_seconds": sampling_seconds, + "median_sampling_seconds": statistics.median(sampling_seconds), + "sampling_steps": [row["sampling_steps"] for row in measured], + "sampling_peak_allocated_bytes": [ + row["sampling_peak_allocated_bytes"] for row in measured + ], + "sampling_peak_reserved_bytes": [ + row["sampling_peak_reserved_bytes"] for row in measured + ], + "latent_checksums": { + "video_sha256": checksum_pairs[0][0], + "audio_sha256": checksum_pairs[0][1], + "exact_across_measured_runs": exact_parity, + "matches_canonical_reference": True, + }, + "fc2_dispatch": { + "required_per_run": 600, + "runs": [row["fc2_dispatch_delta"] for row in measured], + "all_passed": True, + }, + "canonical_warmup": warmup, + "measured_responses": measured, + "ready_before": ready_before, + "ready_after": get_json(f"{args.server}/ready"), + } + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8") + print(json.dumps(report, indent=2), flush=True) + + +if __name__ == "__main__": + main() diff --git a/tools/build_post_fc2_profile_summary.py b/tools/build_post_fc2_profile_summary.py new file mode 100644 index 0000000..5b06172 --- /dev/null +++ b/tools/build_post_fc2_profile_summary.py @@ -0,0 +1,193 @@ +"""Build the post-FC2 authoritative GB10 baseline and profile summary.""" + +from __future__ import annotations + +import argparse +import json +from pathlib import Path + + +def load(path: Path) -> dict: + return json.loads(path.read_text(encoding="utf-8")) + + +def delta(current: float, previous: float) -> dict: + return { + "previous": previous, + "current": current, + "absolute_change": current - previous, + "percent_change": (current / previous - 1.0) * 100.0, + } + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--root", type=Path, default=Path(__file__).resolve().parents[1]) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + benchmarks = args.root / "benchmarks" + resident = load(benchmarks / "gb10-post-fc2-resident-baseline-20260826.json") + block = load(benchmarks / "gb10-post-fc2-block24-profile-20260826.json") + nsys = load(benchmarks / "gb10-post-fc2-warmed-step-nsys-summary-20260826.json") + ncu = load(benchmarks / "gb10-post-fc2-block24-targeted-summary-20260826.json") + previous = load(benchmarks / "gb10-fully-fused-fresh-nsight-summary.json") + + current_components = nsys["components"] + previous_components = previous["one_warmed_sampling_step"]["components"] + component_comparison = { + name: delta( + current_components[name]["milliseconds"], + previous_components[name]["milliseconds"], + ) + for name in current_components + } + ranking = sorted( + ( + { + "component": name, + "milliseconds": values["milliseconds"], + "percent_of_kernel_time": values["percent_of_kernel_time"], + } + for name, values in current_components.items() + ), + key=lambda row: row["milliseconds"], + reverse=True, + ) + previous_step = previous["one_warmed_sampling_step"] + report = { + "name": "gb10-post-fc2-production-profile", + "measurement_date": "2026-08-26", + "source_commit": resident["source_commit"], + "measurement_overlay": ( + "the measurement image included deferred step events and latent hashes plus an opt-in " + "100-token synthetic-text request; the final production overlay gates diagnostics to " + "that benchmark request, and model math is unchanged" + ), + "image": resident["image"], + "device": "NVIDIA GB10", + "compute_capability": "SM121", + "tools": { + "torch": "2.9.1+cu130", + "cuda": "13.0", + "nsight_systems": "2025.3.2.474-253236389321v0", + "nsight_compute": "2025.3.1", + }, + "workload": { + "resolution": [1344, 768], + "frames": 124, + "packed_tokens": 37810, + "text_tokens": 100, + "steps": 12, + "seed": 440420, + "attention": "sage2", + }, + "configuration": { + "H3_NVFP4_SCALE_BACKEND": "vortex", + "H3_NVFP4_SCALE_VERSION": "1", + "H3_FUSED_ELEMENTWISE": "1", + "H3_NVFP4_MODULATE_FUSION": "1", + "H3_NVFP4_SWIGLU_FUSION": "1", + "H3_NVFP4_FC2_LT_SPLITK1": "1", + "H3_SAGE_QKV_LAYOUT": "strided_nhd", + }, + "authoritative_resident_baseline": { + "sampling_seconds": resident["sampling_seconds"], + "median_sampling_seconds": resident["median_sampling_seconds"], + "per_step_seconds": resident["sampling_steps"], + "peak_allocated_bytes": resident["sampling_peak_allocated_bytes"], + "peak_reserved_bytes": resident["sampling_peak_reserved_bytes"], + "latent_checksums": resident["latent_checksums"], + "fc2_dispatch": resident["fc2_dispatch"], + "measurement_policy": resident["measurement_policy"], + }, + "block_24": { + "uninstrumented_module_p50_ms": block["module_forward"]["p50_s"] * 1000.0, + "uninstrumented_module_samples": block["module_forward"]["count"], + "synchronized_decomposition_is_attribution_only": True, + "fc2_schedule_note": ( + "profile_h3_block.py decomposes generic NVFP4 linear calls and bypasses " + "forward_swiglu/guarded FC2; use resident NSYS and targeted NCU for production FC2" + ), + }, + "one_warmed_sampling_step_nsys": { + "elapsed_seconds": load( + benchmarks / "gb10-post-fc2-warmed-step-capture-20260826.json" + )["elapsed_seconds"], + "gpu_span_seconds": nsys["gpu_span_ns"] / 1.0e9, + "kernel_time_seconds": nsys["kernel_time_ns"] / 1.0e9, + "gpu_operation_count": nsys["gpu_operation_count"], + "kernel_count": nsys["kernel_count"], + "kernel_busy_percent_of_span": nsys["kernel_busy_percent_of_span"], + "launch_gaps": nsys["launch_gaps"], + "cpu_gpu_overlap": nsys["cpu_gpu_overlap"], + "components": current_components, + }, + "block_24_targeted_ncu": ncu, + "comparison_to_pre_fc2_profile": { + "previous_summary": "benchmarks/gb10-fully-fused-fresh-nsight-summary.json", + "block_24_p50_ms": delta( + block["module_forward"]["p50_s"] * 1000.0, + previous["block_24"]["uninstrumented_module_p50_ms"], + ), + "block_24_comparison_note": ( + "same-script decomposition control only; it excludes guarded FC2 and must not " + "be interpreted as the production FC2 gain" + ), + "warmed_step_gpu_span_seconds": delta( + nsys["gpu_span_ns"] / 1.0e9, + previous_step["gpu_span_seconds"], + ), + "warmed_step_kernel_time_seconds": delta( + nsys["kernel_time_ns"] / 1.0e9, + previous_step["kernel_time_seconds"], + ), + "warmed_step_kernel_count": delta( + nsys["kernel_count"], previous_step["kernel_launches"], + ), + "components": component_comparison, + }, + "bottleneck_ranking": ranking, + "decision": { + "authoritative_exact_baseline": ( + f"{resident['median_sampling_seconds']:.6f} s median resident sampling" + ), + "time_bottleneck": ( + "Sage2 remains dominant at 62.36% of warmed-step kernel time; its mainloop " + "is 241.91 ms average in NSYS and 259.00 ms in the NCU replay." + ), + "secondary_bottlenecks": ( + "NVFP4 GEMMs are 19.95%, packing 9.70%, norm/RoPE 5.19%, and gate/add 2.63%." + ), + "next_optimization": ( + "None started. Sage2 is the next-ranked investigation target; any implementation " + "requires a separate approved experiment after this baseline is accepted." + ), + }, + "artifacts": [ + "benchmarks/gb10-post-fc2-resident-baseline-20260826.json", + "benchmarks/gb10-post-fc2-block24-profile-20260826.json", + "benchmarks/gb10-post-fc2-warmed-step-capture-20260826.json", + "benchmarks/gb10-post-fc2-warmed-step-20260826.nsys-rep", + "benchmarks/gb10-post-fc2-warmed-step-20260826.sqlite", + "benchmarks/gb10-post-fc2-warmed-step-stats_cuda_gpu_trace.csv", + "benchmarks/gb10-post-fc2-warmed-step-stats_cuda_kern_exec_trace.csv", + "benchmarks/gb10-post-fc2-warmed-step-stats_nvtx_gpu_proj_trace.csv", + "benchmarks/gb10-post-fc2-warmed-step-stats_cuda_gpu_kern_sum.csv", + "benchmarks/gb10-post-fc2-warmed-step-stats_cuda_api_sum.csv", + "benchmarks/gb10-post-fc2-block24-targeted-capture-20260826.json", + "benchmarks/gb10-post-fc2-block24-targeted-20260826.ncu-rep", + "benchmarks/gb10-post-fc2-block24-targeted-20260826.csv", + "benchmarks/gb10-post-fc2-block24-targeted-traffic-capture-20260826.json", + "benchmarks/gb10-post-fc2-block24-targeted-traffic-20260826.ncu-rep", + "benchmarks/gb10-post-fc2-block24-targeted-traffic-20260826.csv", + "benchmarks/gb10-post-fc2-warmed-step-nsys-summary-20260826.json", + "benchmarks/gb10-post-fc2-block24-targeted-summary-20260826.json", + ], + } + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8") + print(json.dumps(report, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/tools/profile_sampling_stages.py b/tools/profile_sampling_stages.py index 64a17a3..badf9d6 100644 --- a/tools/profile_sampling_stages.py +++ b/tools/profile_sampling_stages.py @@ -3,8 +3,11 @@ from __future__ import annotations import argparse +import hashlib import inspect import json +import os +import sys import time from pathlib import Path @@ -13,6 +16,7 @@ import torch from h3_blackwell_runtime.attention import AVAILABLE_BACKENDS from h3_blackwell_runtime.checkpoint import H3Checkpoint from h3_blackwell_runtime.denoiser import H3PackedDenoiser +from h3_blackwell_runtime.fc2_lt import fc2_lt_status, prepare_fc2_lt from h3_blackwell_runtime.packing import H3PromptPacker from h3_blackwell_runtime.sampler import sample_video_res_multistep from h3_blackwell_runtime.t2v import random_av_latents @@ -40,6 +44,11 @@ def parse_args() -> argparse.Namespace: return parser.parse_args() +def tensor_sha256(value: torch.Tensor) -> str: + immutable = value.detach().contiguous().view(torch.uint16).cpu() + return hashlib.sha256(immutable.numpy().tobytes()).hexdigest() + + def main() -> None: args = parse_args() torch.manual_seed(args.seed) @@ -47,6 +56,8 @@ def main() -> None: model = H3PackedDenoiser.from_checkpoint( checkpoint, output_dtype=torch.bfloat16, attention_backend=args.attention, ).eval() + if not prepare_fc2_lt(): + raise RuntimeError("guarded FC2 extension preparation failed") packer = H3PromptPacker(checkpoint) if hasattr(checkpoint, "release_cache"): checkpoint.release_cache() @@ -75,6 +86,7 @@ def main() -> None: torch.cuda.synchronize() if args.cuda_profiler_capture: torch.cuda.cudart().cudaProfilerStart() + fc2_before = fc2_lt_status() started = time.perf_counter() supports_stage_trace = "sampling_stage_trace" in inspect.signature( sample_video_res_multistep, @@ -99,6 +111,9 @@ def main() -> None: elapsed = time.perf_counter() - started if args.cuda_profiler_capture: torch.cuda.cudart().cudaProfilerStop() + fc2_after = fc2_lt_status() + peak_allocated = torch.cuda.max_memory_allocated() + peak_reserved = torch.cuda.max_memory_reserved() report = { "device": torch.cuda.get_device_name(), "torch": torch.__version__, @@ -110,11 +125,24 @@ def main() -> None: "text_tokens": args.text_tokens, "warmup_runs": args.warmup_runs, "cuda_profiler_capture": args.cuda_profiler_capture, + "argv": sys.argv, + "environment_switches": { + name: value for name, value in sorted(os.environ.items()) + if name.startswith(("H3_", "CUDA_", "TORCH_")) + }, "elapsed_seconds": elapsed, "stage_trace": trace, "checksums": [sampled_video.float().sum().item(), sampled_audio.float().sum().item()], - "peak_allocated_bytes": torch.cuda.max_memory_allocated(), - "peak_reserved_bytes": torch.cuda.max_memory_reserved(), + "sha256": { + "video": tensor_sha256(sampled_video), + "audio": tensor_sha256(sampled_audio), + }, + "fc2_dispatch_delta": { + name: fc2_after[name] - fc2_before[name] + for name in ("attempts", "successes", "fallbacks") + }, + "peak_allocated_bytes": peak_allocated, + "peak_reserved_bytes": peak_reserved, "measurement_policy": ( "uninstrumented sampling wall time" if args.uninstrumented or not supports_stage_trace diff --git a/tools/serve_hot_runtime.py b/tools/serve_hot_runtime.py index adf6201..2fe6e66 100644 --- a/tools/serve_hot_runtime.py +++ b/tools/serve_hot_runtime.py @@ -164,6 +164,12 @@ class Handler(BaseHTTPRequestHandler): cache_start_percent = float(payload.get("cache_start_percent", 0.0)) cache_end_percent = float(payload.get("cache_end_percent", 1.0)) cache_subsample_factor = int(payload.get("cache_subsample_factor", 2)) + benchmark_text_tokens = payload.get("benchmark_text_tokens") + if benchmark_text_tokens is not None: + benchmark_text_tokens = int(benchmark_text_tokens) + if benchmark_text_tokens != 100 or first_frame is not None or last_frame is not None: + write_json(self, 400, {"error": "benchmark_text_tokens requires canonical T2V value 100"}) + return started = time.perf_counter() with runtime_lock: result = runtime.generate( @@ -188,6 +194,7 @@ class Handler(BaseHTTPRequestHandler): cache_start_percent=cache_start_percent, cache_end_percent=cache_end_percent, cache_subsample_factor=cache_subsample_factor, + benchmark_text_tokens=benchmark_text_tokens, ) result["wall_seconds"] = time.perf_counter() - started write_json(self, 200, result) diff --git a/tools/summarize_ncu_profile.py b/tools/summarize_ncu_profile.py new file mode 100644 index 0000000..227d78a --- /dev/null +++ b/tools/summarize_ncu_profile.py @@ -0,0 +1,127 @@ +"""Extract targeted H3 block metrics from an Nsight Compute raw CSV export.""" + +from __future__ import annotations + +import argparse +import csv +import json +from pathlib import Path + + +ROLES = ("qkv", "sage2", "attention_output", "fc1", "fc2") +EXPECTED_GRIDS = ( + "(296, 168, 1)", + "(296, 56, 1)", + "(296, 42, 1)", + "(296, 224, 1)", + "(296, 42, 1)", +) + + +def number(row: dict[str, str], name: str) -> float | None: + value = row.get(name, "") + if value in {"", "no data", "n/a"}: + return None + return float(value) + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--input", type=Path, required=True) + parser.add_argument("--traffic", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + + with args.input.open(newline="", encoding="utf-8") as handle: + reader = csv.DictReader(handle) + units = next(reader) + rows = list(reader) + if len(rows) != len(ROLES): + raise RuntimeError(f"expected five targeted launches, found {len(rows)}") + with args.traffic.open(newline="", encoding="utf-8") as handle: + traffic_reader = csv.DictReader(handle) + next(traffic_reader) + traffic_rows = list(traffic_reader) + if len(traffic_rows) != len(ROLES): + raise RuntimeError(f"expected five traffic launches, found {len(traffic_rows)}") + for index, (role, row, traffic_row, expected_grid) in enumerate( + zip(ROLES, rows, traffic_rows, EXPECTED_GRIDS, strict=True) + ): + if int(row["ID"]) != index or int(traffic_row["ID"]) != index: + raise RuntimeError(f"{role} launch ID/order contract failed") + if row["Kernel Name"] != traffic_row["Kernel Name"]: + raise RuntimeError(f"{role} kernel differs between metric and traffic passes") + if row["Grid Size"] != expected_grid or traffic_row["Grid Size"] != expected_grid: + raise RuntimeError(f"{role} grid/order contract failed") + if role == "sage2" and "qk_int_sv_f8_attn_kernel" not in row["Kernel Name"]: + raise RuntimeError("Sage2 launch contract failed") + if role != "sage2" and "block_scaled" not in row["Kernel Name"]: + raise RuntimeError(f"{role} NVFP4 GEMM launch contract failed") + + stall_prefix = "smsp__average_warps_issue_stalled_" + stall_suffix = "_per_issue_active.ratio" + output = {} + for role, row, traffic_row in zip(ROLES, rows, traffic_rows, strict=True): + stalls = [] + for name in row: + if name.startswith(stall_prefix) and name.endswith(stall_suffix): + value = number(row, name) + if value is not None: + stalls.append({ + "reason": name[len(stall_prefix):-len(stall_suffix)], + "warps_per_issue_active": value, + }) + stalls.sort(key=lambda item: item["warps_per_issue_active"], reverse=True) + output[role] = { + "launch_id": int(row["ID"]), + "kernel_name": row["Kernel Name"], + "grid_size": row["Grid Size"], + "block_size": row["Block Size"], + "duration_ns": number(row, "gpu__time_duration.sum"), + "registers_per_thread": number(row, "launch__registers_per_thread"), + "achieved_occupancy_percent": number( + row, "sm__warps_active.avg.pct_of_peak_sustained_active" + ), + "eligible_warps_per_scheduler": number( + row, "smsp__warps_eligible.avg.per_cycle_active" + ), + "issue_active_percent": number( + row, "smsp__issue_active.avg.pct_of_peak_sustained_active" + ), + "sm_throughput_percent": number( + row, "sm__throughput.avg.pct_of_peak_sustained_elapsed" + ), + "tensor_pipe_active_percent": number( + row, "sm__pipe_tensor_cycles_active.avg.pct_of_peak_sustained_elapsed" + ), + "l2_requested_bytes": number(traffic_row, "lts__t_bytes.sum"), + "l2_hit_rate_percent": number(traffic_row, "lts__t_sector_hit_rate.pct"), + "l2_throughput_percent": number( + row, "lts__throughput.avg.pct_of_peak_sustained_elapsed" + ), + "memory_throughput_percent": number( + traffic_row, "gpu__compute_memory_throughput.avg.pct_of_peak_sustained_elapsed" + ), + "local_spilling_requests": number(row, "derived__local_spilling_requests"), + "top_scheduler_stalls": stalls[:5], + } + + report = { + "source_csv": str(args.input), + "source_traffic_csv": str(args.traffic), + "launch_order_contract": list(ROLES), + "cache_control": "none (warmed/uncontrolled cache, as reported by NCU)", + "metrics": output, + "units": { + "duration_ns": "ns", + "l2_requested_bytes": "lts__t_bytes.sum", + "throughput_and_hit_rate": "%", + }, + } + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8") + print(json.dumps(report, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/tools/summarize_nsys_profile.py b/tools/summarize_nsys_profile.py new file mode 100644 index 0000000..510a4d6 --- /dev/null +++ b/tools/summarize_nsys_profile.py @@ -0,0 +1,166 @@ +"""Summarize an Nsight Systems CUDA trace into stable H3 component categories.""" + +from __future__ import annotations + +import argparse +import csv +import json +from pathlib import Path + + +def category(name: str) -> str: + if any(token in name for token in ( + "qk_int_sv_f8_attn_kernel", + "MeanScaleKernel", + "TransposePadPermuteKernel", + "QuantInt8Kernel", + )): + return "sage2" + if "cutlass3x_sm120_bstensorop" in name: + return "nvfp4_gemms" + if any(token in name for token in ( + "partial_absmax_", + "quantize_nvfp4_", + "quantize_nvfp4_kernel", + "final_scale_", + "FillFunctor", + )): + return "nvfp4_packing" + if any(token in name for token in ( + "rope_kernel", + "vectorized_layer_norm_kernel", + "MeanOps list[tuple[int, int]]: + output: list[list[int]] = [] + for start, end in sorted(intervals): + if not output or start > output[-1][1]: + output.append([start, end]) + else: + output[-1][1] = max(output[-1][1], end) + return [(start, end) for start, end in output] + + +def total(intervals: list[tuple[int, int]]) -> int: + return sum(end - start for start, end in merge(intervals)) + + +def intersection_total( + left: list[tuple[int, int]], right: list[tuple[int, int]], +) -> int: + left = merge(left) + right = merge(right) + i = j = result = 0 + while i < len(left) and j < len(right): + start = max(left[i][0], right[j][0]) + end = min(left[i][1], right[j][1]) + result += max(0, end - start) + if left[i][1] <= right[j][1]: + i += 1 + else: + j += 1 + return result + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--gpu-trace", type=Path, required=True) + parser.add_argument("--kernel-exec-trace", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + + with args.gpu_trace.open(newline="", encoding="utf-8") as handle: + gpu_rows = list(csv.DictReader(handle)) + kernels = [row for row in gpu_rows if row["GrdX"]] + gpu_intervals = [ + (int(row["Start (ns)"]), int(row["Start (ns)"]) + int(row["Duration (ns)"])) + for row in kernels + ] + all_gpu_intervals = [ + (int(row["Start (ns)"]), int(row["Start (ns)"]) + int(row["Duration (ns)"])) + for row in gpu_rows + ] + first_gpu = min(start for start, _ in all_gpu_intervals) + last_gpu = max(end for _, end in all_gpu_intervals) + kernel_time = sum(int(row["Duration (ns)"]) for row in kernels) + component_ns = { + name: 0 for name in ( + "sage2", "nvfp4_gemms", "nvfp4_packing", "norm_and_rope", + "remaining_gate_add", "other", + ) + } + component_launches = component_ns.copy() + for row in kernels: + name = category(row["Name"]) + component_ns[name] += int(row["Duration (ns)"]) + component_launches[name] += 1 + + ordered = sorted(gpu_intervals) + gaps = [ + max(0, ordered[index][0] - ordered[index - 1][1]) + for index in range(1, len(ordered)) + ] + positive_gaps = [gap for gap in gaps if gap] + + with args.kernel_exec_trace.open(newline="", encoding="utf-8") as handle: + launch_rows = list(csv.DictReader(handle)) + api_intervals = [ + (int(row["API Start (ns)"]), int(row["API Start (ns)"]) + int(row["API Dur (ns)"])) + for row in launch_rows + ] + launch_api_time = total(api_intervals) + launch_api_gpu_overlap = intersection_total(api_intervals, gpu_intervals) + gpu_span = last_gpu - first_gpu + + report = { + "source_gpu_trace": str(args.gpu_trace), + "source_kernel_exec_trace": str(args.kernel_exec_trace), + "gpu_span_ns": gpu_span, + "gpu_operation_count": len(gpu_rows), + "kernel_count": len(kernels), + "kernel_time_ns": kernel_time, + "kernel_busy_percent_of_span": kernel_time / gpu_span * 100.0, + "launch_gaps": { + "positive_gap_count": len(positive_gaps), + "total_ns": sum(positive_gaps), + "average_ns": sum(positive_gaps) / len(positive_gaps) if positive_gaps else 0, + "maximum_ns": max(positive_gaps, default=0), + }, + "cpu_gpu_overlap": { + "scope": "CUDA kernel-launch API intervals intersected with GPU kernel intervals", + "launch_api_union_ns": launch_api_time, + "launch_api_gpu_overlap_ns": launch_api_gpu_overlap, + "launch_api_overlap_percent": ( + launch_api_gpu_overlap / launch_api_time * 100.0 if launch_api_time else 0 + ), + }, + "components": { + name: { + "milliseconds": value / 1.0e6, + "percent_of_kernel_time": value / kernel_time * 100.0, + "launches": component_launches[name], + } + for name, value in component_ns.items() + }, + "classification_policy": { + "sage2": "mainloop plus MeanScale/TransposePadPermute/QuantInt8 preparation", + "nvfp4_gemms": "SM120 block-scaled CUTLASS GEMMs", + "nvfp4_packing": "absmax, final-scale, zero-fill and NVFP4 quantization kernels", + "norm_and_rope": "layer norm, mean reduction and fused RMS/RoPE kernels", + "remaining_gate_add": "fused residual gate/add kernels", + "other": "all unmatched kernels", + }, + } + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8") + print(json.dumps(report, indent=2)) + + +if __name__ == "__main__": + main()