↓ 1 callersMethodcheck_context(self, bs, seq, q_nheads, kv_nheads, k_head_dim, v_head_dim)
python/triton_dist/kernels/nvidia/ulysses_sp_dispatch.py:528
↓ 1 callersFunctionchunk_gated_delta_rule_fwd(
q: torch.Tensor,
k: torch.Tensor,
v: torch.Tensor,
g: torch.Tensor,
beta: torch.Tensor,
python/triton_dist/kernels/nvidia/gdn.py:926
↓ 1 callersFunctionchunk_kkt_inv_ut_fused_triton varlen mode k: [1, T, H, K] v: [1, T, H, V] beta: [1, T, H] g: [1, T, H] cu_seqlens: [B + 1] chunk_indices: [NT * 2] wher
python/triton_dist/kernels/nvidia/gdn.py:308
↓ 1 callersFunctionconsumer_all_reduce(symm_buf, tile_signal, BLOCK_SIZE_M=16, BLOCK_SIZE_N=64, GROUP_SIZE_M=1, NUM_COMM_SMS=16)
python/triton_dist/kernels/amd/gemm_allreduce.py:337
↓ 1 callersFunctionconsumer_all_reduce_kernel(
symm_input_ptr,
symm_ar_out_ptr,
ar_out_ptr, #
gemm_barrier_ptr,
multi_st_barrier_ptr,
python/triton_dist/kernels/nvidia/gemm_allreduce.py:140
↓ 1 callersFunctioncopy_kernel(
rank,
local_buf_ptr,
global_buf_ptr,
M_per_rank,
N,
stride_local_m,
stride_local
python/triton_dist/kernels/nvidia/allgather_gemm.py:48
↓ 1 callersFunctioncp_engine_producer_all_gather_put(local_tensor, ag_buffer, signal_buffer,
M_per_rank, N, signal_target, r
tutorials/07-overlapping-allgather-gemm.py:308
↓ 1 callersFunctioncreate_all_to_all_context(
max_m: int,
hidden: int,
rank: int,
num_tot_experts: int,
WORLD_SIZE: int,
experts_p
python/triton_dist/kernels/amd/low_latency_all_to_all.py:193
↓ 1 callersFunctioncreate_all_to_all_single_2d_context(
max_m: int,
hidden_dim: int,
rank: int,
world_size: int,
dtype=torch.bfloat16,
)
python/triton_dist/kernels/nvidia/all_to_all_single_2d.py:145
↓ 1 callersFunctioncreate_ll_gemm_ar_context(rank, world_size, local_world_size, max_M, N, dtype, MIN_BLOCK_SIZE_M=16,
MIN_B
python/triton_dist/kernels/nvidia/gemm_allreduce.py:127
↓ 1 callersFunctioncreate_moe_ar_context(rank, world_size, local_world_size, max_token_num, hidden_dim, num_experts, topk, input_dtype,
python/triton_dist/kernels/nvidia/moe_reduce_ar.py:316
↓ 1 callersFunctioncreate_sp_ag_attention_context_inter_node(
batch_size,
q_head,
kv_head,
max_seqlen_k,
max_q_shard_len,
head_dim,
input_dtyp
python/triton_dist/kernels/nvidia/sp_ag_attention_inter_node.py:57
↓ 1 callersFunctioncreate_sp_ag_attention_context_intra_node(
batch_size,
q_head,
kv_head,
max_seqlen_k,
max_q_shard_len,
head_dim,
input_dtyp
python/triton_dist/kernels/nvidia/sp_ag_attention_intra_node.py:60
↓ 1 callersFunctioncreate_ulysses_sp_pre_attn_comm_context(bs: int, max_seq: int, q_nheads: int, k_head_dim: int, v_head_dim: int,
python/triton_dist/kernels/nvidia/ulysses_sp_dispatch.py:546