Tue Sep 15 14:35:55 2026    /tmp/dispatch_profile_rank0.pstats

         25989600 function calls (25715773 primitive calls) in 26.225 seconds

   Ordered by: internal time
   List reduced from 1705 to 40 due to restriction <40>

   ncalls  tottime  percall  cumtime  percall filename:lineno(function)
    44160   10.836    0.000   10.964    0.000 /opt/venv/lib/python3.12/site-packages/torch/_library/custom_ops.py:374(backend_impl)
 9598/624    0.615    0.000    0.772    0.001 /opt/venv/lib/python3.12/site-packages/torch/_functorch/_aot_autograd/runtime_wrappers.py:489(run)
    88320    0.430    0.000    0.430    0.000 {_launch_kernel}
    51840    0.416    0.000    1.443    0.000 /opt/venv/lib/python3.12/site-packages/triton/runtime/jit.py:708(run)
    40320    0.411    0.000    1.681    0.000 /opt/venv/lib/python3.12/site-packages/triton/runtime/autotuner.py:212(run)
    42240    0.343    0.000   35.748    0.001 /opt/venv/lib/python3.12/site-packages/torch/_ops.py:870(__call__)
    88320    0.332    0.000    1.184    0.000 /opt/venv/lib/python3.12/site-packages/torch/_inductor/runtime/triton_heuristics.py:1650(run)
    51840    0.294    0.000    0.324    0.000 {built-in method cuda_utils.launch}
   951726    0.284    0.000    0.809    0.000 <frozen os>:680(__getitem__)
    40320    0.277    0.000    0.364    0.000 /opt/venv/lib/python3.12/site-packages/fla/ops/utils/cache.py:158(build)
       16    0.275    0.017    0.389    0.024 /opt/venv/lib/python3.12/site-packages/tkv/kernels/cuda/prefill/turbo_attn_int8.py:263(best_int8_unit)
    48000    0.266    0.000    0.266    0.000 {built-in method arbi_serve_exl3_i8_gemm_v1.gemm}
    77016    0.265    0.000    0.267    0.000 {built-in method torch.empty}
136316/625    0.248    0.000    0.688    0.001 /opt/venv/lib/python3.12/site-packages/torch/_dynamo/eval_frame.py:1275(_fn)
  1888216    0.215    0.000    0.399    0.000 <frozen os>:766(decode)
    48000    0.203    0.000    0.413    0.000 /opt/venv/lib/python3.12/site-packages/arbi_serve/weight_quant/exl3/custom_op.py:2391(_int8_leg_serves)
   955566    0.190    0.000    0.323    0.000 <frozen os>:762(encode)
   956614    0.189    0.000    1.319    0.000 <frozen _collections_abc>:892(__iter__)
  1888558    0.184    0.000    0.184    0.000 {method 'decode' of 'bytes' objects}
    17280    0.182    0.000    8.679    0.001 /opt/venv/lib/python3.12/site-packages/torch/_library/custom_ops.py:682(adinplaceorview_impl)
    79680    0.182    0.000    1.596    0.000 /opt/venv/lib/python3.12/site-packages/tkv/config.py:889(<genexpr>)
    96000    0.180    0.000    0.189    0.000 {method 'narrow' of 'torch._C.TensorBase' objects}
670329/668879    0.164    0.000    0.215    0.000 {built-in method builtins.getattr}
    88320    0.157    0.000    0.618    0.000 /opt/venv/lib/python3.12/site-packages/torch/_inductor/runtime/static_triton_launcher.py:232(run)
     9600    0.148    0.000    0.878    0.000 /opt/venv/lib/python3.12/site-packages/arbi_serve/weight_quant/exl3/prologue.py:538(_add_rmsnorm_group)
   589078    0.146    0.000    0.146    0.000 {method 'get' of 'dict' objects}
   957600    0.146    0.000    0.341    0.000 <frozen os>:703(__iter__)
 9596/100    0.144    0.000    0.148    0.001 /opt/venv/lib/python3.12/site-packages/torch/_functorch/aot_autograd.py:1265(forward)
     1920    0.143    0.000    0.929    0.000 /opt/venv/lib/python3.12/site-packages/arbi_serve/models/_qwen3_5_layers.py:169(_pre_attn)
   360960    0.136    0.000    0.136    0.000 {built-in method torch._C._dynamo.guards.assert_size_stride}
     5760    0.133    0.000    0.134    0.000 {built-in method torch.mm}
    28800    0.126    0.000    1.728    0.000 /opt/venv/lib/python3.12/site-packages/triton/runtime/autotuner.py:456(run)
  1448409    0.121    0.000    0.131    0.000 {built-in method builtins.hasattr}
    65792    0.121    0.000    0.122    0.000 {method 'contiguous' of 'torch._C.TensorBase' objects}
1782251/1773571    0.118    0.000    0.127    0.000 {built-in method builtins.isinstance}
    29760    0.118    0.000    0.119    0.000 {method 'to' of 'torch._C.TensorBase' objects}
    86880    0.116    0.000    0.119    0.000 {method 'view' of 'torch._C.TensorBase' objects}
   437760    0.114    0.000    0.114    0.000 {built-in method triton._C.libtriton.native_specialize_impl}
    82560    0.109    0.000    0.109    0.000 {built-in method torch._C._dynamo.guards._empty_strided_cuda}
    42224    0.106    0.000    0.109    0.000 {built-in method torch.empty_like}


