MCPcopy Create free account

hub / github.com/ByteDance-Seed/Triton-distributed / functions

Functions5,103 in github.com/ByteDance-Seed/Triton-distributed

↓ 2 callersMethodget_task_type_id
(cls)
python/triton_dist/mega_triton_kernel/core/task_base.py:181
↓ 2 callersMethodget_tensor_breakdown
Get a breakdown of memory usage by tensor. Returns: Dict mapping tensor name to size in bytes
python/triton_dist/utils.py:1405
↓ 2 callersFunctionget_tensorcore_tflops_by_calc
return compute throughput in TOPS
python/triton_dist/kernels/nvidia/gemm_perf_model.py:96
↓ 2 callersFunctionget_tensorcore_tflops_by_device_name
TFLOPS with no sparse.
python/triton_dist/kernels/nvidia/gemm_perf_model.py:144
↓ 2 callersFunctionget_thirdparty_packages
(packages: list)
python/setup.py:379
↓ 2 callersFunctionget_triton_cache_path
()
python/setup.py:352
↓ 2 callersFunctionget_triton_dist_moe_profile_enabled
Get current Ditron MoE profiling state. Returns: Dictionary with profiling flags for each operation.
python/triton_dist/function/nvidia/common.py:120
↓ 2 callersFunctionget_triton_runner
(profiler_buffer)
python/triton_dist/kernels/amd/gemm.py:838
↓ 2 callersFunctionget_triton_split_kv_algo_info
(q_heads, kv_heads, q_head_dim, v_head_dim, page_size, split_kv=32, soft_cap=0.0)
python/triton_dist/kernels/nvidia/flash_decode.py:103
↓ 2 callersFunctionget_triton_subpackages
Walk a triton subdirectory to enumerate all subpackages dynamically.
python/setup.py:1059
↓ 2 callersMethodget_type
Get the LLType of a value.
python/little_kernel/codegen/codegen_base.py:223
↓ 2 callersMethodget_underlying_tensor
Get the underlying tensor, or None if not materialized.
python/triton_dist/utils.py:1179
↓ 2 callersFunctiongqa_fwd_batch_decode
(q, k_cache, v_cache, workspace, q_lens, kv_lens, block_table, scale, soft_cap=0.0, o
python/triton_dist/kernels/nvidia/flash_decode.py:763
↓ 2 callersFunctiongqa_fwd_batch_decode_aot
(stream, q, k_cache, v_cache, workspace, q_lens, kv_lens, block_table, scale, soft_cap=0,
python/triton_dist/kernels/nvidia/flash_decode.py:979
↓ 2 callersFunctiongqa_fwd_batch_decode_persistent
(q, k_cache, v_cache, workspace, q_lens, kv_lens, block_table, scale, soft_cap=0,
python/triton_dist/kernels/nvidia/flash_decode.py:931
↓ 2 callersFunctiongqa_fwd_batch_decode_persistent_aot
(stream, q, k_cache, v_cache, workspace, q_lens, kv_lens, block_table, scale,
python/triton_dist/kernels/nvidia/flash_decode.py:1095
↓ 2 callersFunctiongroup_gemm
scatter_idx: [M, topk]
python/triton_dist/test/nvidia/test_ep_moe_inference.py:235
↓ 2 callersFunctiongroupgemm_torch
( input: torch.Tensor, weight: torch.Tensor, scatter_indices: torch.Tensor, exp_indices: torch
python/triton_dist/test/nvidia/test_ep_moe_inference.py:193
↓ 2 callersFunctionhistogram_block_kernel
(values_ptr, N, BLOCK_SIZE: tl.constexpr, NUM_BINS: tl.constexpr)
python/triton_dist/kernels/nvidia/moe_utils.py:97
↓ 2 callersFunctioninit_triton_dist_ep_op
(ep_group, max_tokens_per_rank, hidden_size, topk, ep_rank, num_experts, ep_size, d
python/triton_dist/function/nvidia/common.py:173
↓ 2 callersMethodis_const
(self)
python/little_kernel/core/type_system.py:271
↓ 2 callersFunctionis_cuda
()
python/triton_dist/kernels/nvidia/gemm.py:39
↓ 2 callersFunctionis_empty_slot
(val)
python/triton_dist/tools/profiler/context.py:61
↓ 2 callersFunctionis_git_repo
Return True if this file resides in a git repository
python/setup.py:47
↓ 2 callersMethodis_grid_constant
Check if the type is a grid constant
python/little_kernel/core/type_system.py:51
↓ 2 callersMethodis_int
(self)
python/triton_dist/tools/tune/tune_gemm.py:210
↓ 2 callersMethodis_modifier_call
Check if a Call node represents a loop modifier. Args: node: AST Call node to check ctx: Context dic
python/little_kernel/core/passes/utils/registries/loop_modifier_registry.py:89
↓ 2 callersMethodis_scalar
Check if the type is a scalar
python/little_kernel/core/type_system.py:39
↓ 2 callersFunctionis_struct_stub
Check if an object is a struct stub.
python/little_kernel/language/intrin/struct_stub.py:233
↓ 2 callersFunctionissue_meta_prefetch
( meta_mbar_full: ll.ptr[ll.uint64], # &meta_mbar_full[buf] meta_send_mask: ll.ptr[ll.int32], # &met
python/little_kernel/design/flashcomm_dispatch_chunk.py:62
↓ 2 callersFunctionkernel_ring_reduce_non_tma
( in_ptr, # c of shape [NUM_SPLITS, elems_per_rank] out_ptr, # out = sum(c, axis=0) of shape [elems_
python/triton_dist/kernels/nvidia/reduce_scatter.py:639
↓ 2 callersFunctionload
(ptr, scope="agent", semantic="monotonic", _semantic=None)
python/triton_dist/language/extra/hip/language_extra.py:325
↓ 2 callersFunctionload_envreg
(val: tl.constexpr)
python/triton_dist/kernels/amd/common_ops.py:87
↓ 2 callersFunctionload_envreg
(val: tl.constexpr)
python/triton_dist/kernels/nvidia/common_ops.py:89
↓ 2 callersFunctionload_v2_b64
(ptr, _semantic=None)
python/triton_dist/language/extra/cuda/language_extra.py:196
↓ 2 callersFunctionlocal_copy_and_barrier_all
(rank, num_ranks, local_data, global_data, comm_buf, barrier_ptr, M_per_rank, N,
python/triton_dist/kernels/metax/allgather_gemm.py:607
↓ 2 callersMethodlocal_kv_nheads
(self, q_nheads)
python/triton_dist/kernels/nvidia/ulysses_sp_dispatch.py:524
↓ 2 callersMethodmake_barrier_all_intra_node
`wait_inputs` is used to build data dependency.
python/triton_dist/mega_triton_kernel/models/model_builder.py:463
↓ 2 callersFunctionmake_cuda_graph
(mempool, func)
python/triton_dist/test/nvidia/test_tp_attn.py:77
↓ 2 callersFunctionmake_cuda_graph
(mempool, func)
python/triton_dist/test/nvidia/test_tp_e2e.py:87
↓ 2 callersFunctionmake_cuda_graph
(mempool, func)
python/triton_dist/test/nvidia/test_tp_mlp.py:58
↓ 2 callersFunctionmake_cuda_graph
(mempool, func)
python/triton_dist/mega_triton_kernel/test/models/bench_qwen3.py:46
↓ 2 callersFunctionmake_data
(M, N, K, dtype: torch.dtype, trans_b, tp_group: torch.distributed.ProcessGroup)
python/triton_dist/benchmark/bench_allgather_gemm.py:68
↓ 2 callersMethodmake_fc1
(self, input: torch.Tensor, weight: torch.Tensor, output: torch.Tensor, layer_id: int = 0)
python/triton_dist/mega_triton_kernel/models/model_builder.py:229
↓ 2 callersMethodmake_fc2
(self, input: torch.Tensor, weight: torch.Tensor, output: torch.Tensor, layer_id: int = 0)
python/triton_dist/mega_triton_kernel/models/model_builder.py:232
↓ 2 callersMethodmake_flash_attn
Args: q: (bs, seq, nheads_q, head_dim) k: (bs, seq, nheads_kv, head_dim) v: (bs, seq, nhe
python/triton_dist/mega_triton_kernel/models/model_builder.py:310
↓ 2 callersMethodmake_linear
(self, input: torch.Tensor, weight: torch.Tensor, output: torch.Tensor, layer_id: int = 0)
python/triton_dist/mega_triton_kernel/models/model_builder.py:241
↓ 2 callersMethodmake_o_proj
(self, input: torch.Tensor, weight: torch.Tensor, output: torch.Tensor, layer_id: int = 0)
python/triton_dist/mega_triton_kernel/models/model_builder.py:238
↓ 2 callersFunctionmake_ptr_tensor
Create an int64 device tensor holding data_ptr() values (simulating int32_t**).
python/little_kernel/design/test_flashcomm_compute.py:274
↓ 2 callersMethodmake_qk_norm_rope_update_kvcache
this op assume that kv_lens has been update (kv_lens = history_kv_len + seq_len(qkv.shape[1])) inplace update new kv to key_c
python/triton_dist/mega_triton_kernel/models/model_builder.py:344
↓ 2 callersMethodmake_qkv_pack_qk_norm_rope_split_v
Inputs: qkv: [bs, seq_len, num_q_heads + 2 * num_kv_heads, head_dim] kv_lens: [bs] q/k_rm
python/triton_dist/mega_triton_kernel/models/model_builder.py:387
↓ 2 callersMethodmake_silu_mul_up
(self, fc1_out, act_out, layer_id=0)
python/triton_dist/mega_triton_kernel/models/model_builder.py:336
↓ 2 callersFunctionmake_weights
Full expert tables (identical on every rank via a fixed seed) + this rank's local slice. AMD layout: ``W1=[E, hidden, inter]`` (up), ``W2=[E, int
python/triton_dist/test/amd/test_ep_moe_fused.py:128
↓ 2 callersMethodmark_global
Mark a name as global/nonlocal in the current scope. This means the name refers to the outer scope, not a local definition.
python/little_kernel/core/passes/utils/scope_manager.py:195
↓ 2 callersFunctionmatmul
(a, b, activation="")
python/triton_dist/test/amd/test_matmul_amd.py:647
↓ 2 callersFunctionmeasure_cuda_function_performance
(cuda_function, warmup=10, repeat=100)
python/triton_dist/test/nvidia/test_distributed_wait.py:246
↓ 2 callersFunctionmeasure_cuda_function_performance
(cuda_function, warmup=10, repeat=100)
python/triton_dist/test/metax/test_distributed_wait.py:223
↓ 2 callersMethodmega_preprocess_group_gemm
( self, gemm_input_data: torch.Tensor, gemm_weight: torch.Tensor, gemm_expert_
python/triton_dist/layers/nvidia/ep_a2a_fused_layer.py:494
↓ 2 callersFunctionmemcpy_async_kernel
(src_ptr, dst_ptr, N, BLOCK_SIZE: tl.constexpr)
python/triton_dist/kernels/amd/memcpy.py:32
↓ 2 callersFunctionmoe_grouped_gemm
( input_data, weight, expert_ids, split_size, split_size_cum, tile_num, tile_num_c
python/triton_dist/kernels/nvidia/group_gemm.py:728
↓ 2 callersFunctionmultimem_st_v2
(ptr, val0, val1, _semantic=None)
python/triton_dist/language/extra/cuda/language_extra.py:305
↓ 2 callersFunctionmultimem_st_v4
(ptr, val0, val1, val2, val3, _semantic=None)
python/triton_dist/language/extra/cuda/language_extra.py:337
↓ 2 callersFunctionop
()
python/triton_dist/test/nvidia/test_all_to_all_vdev_2d_offset.py:379
↓ 2 callersFunctionopen_url
(url)
python/setup.py:339
↓ 2 callersFunctionpack_f32_bf16x2
(vec, _semantic=None)
python/triton_dist/kernels/nvidia/memory_ops.py:231
↓ 2 callersFunctionparse_output
Parse command output to extract performance data
python/triton_dist/benchmark/bench_tp_attn.py:98
↓ 2 callersFunctionperf_ag
(func, ag_buffers: torch.Tensor, nbytes: int, ctx: AllGatherContext)
tutorials/03-inter-node-allgather.py:196
↓ 2 callersFunctionperf_torch
(sp_group: torch.distributed.ProcessGroup, q: torch.Tensor, k: torch.Tensor, v: torch.Tensor, s
python/triton_dist/test/nvidia/test_ulysses_sp_dispatch.py:96
↓ 2 callersFunctionpersistent_gemm_notify
(a, b, out, gemm_barrier, tile_barrier, gemm_config: triton.Config, use_tma=False,
python/triton_dist/kernels/nvidia/gemm_allreduce.py:626
↓ 2 callersMethodpop_scope
Remove the active scope (e.g., when exiting a function/loop)
python/little_kernel/core/passes/utils/type_inference/infer_type.py:83
↓ 2 callersMethodpost_attn_a2a
( self, inputs: torch.Tensor, seq_lens_cpu: Optional[torch.Tensor] = None, ret
python/triton_dist/kernels/nvidia/sp_ulysess_o_all2all_gemm.py:763
↓ 2 callersMethodpre_attn_a2a
( self, inputs: torch.Tensor, seq_lens_cpu: Optional[torch.Tensor] = None, num
python/triton_dist/kernels/nvidia/sp_ulysess_qkv_gemm_all2all.py:818
↓ 2 callersMethodprepare
(backend_name: str, backend_src_dir: str = None, is_external: bool = False)
python/setup.py:113
↓ 2 callersFunctionprepare_inputs
Prepare inputs for MoE forward pass. Args: ffn_dim: FFN intermediate dimension hidden_dim: Hidden dimension topk
python/triton_dist/test/nvidia/test_ep_moe_fused.py:60
↓ 2 callersFunctionprepare_lens
(cu_seqlens: torch.IntTensor)
python/triton_dist/kernels/nvidia/gdn.py:51
↓ 2 callersMethodpreprocess
Cross-rank split exchange + recv-offset + grouped-GEMM tiling. Built on-device by default (``use_device_metadata=True``); set it False to use
python/triton_dist/layers/amd/ep_a2a_fused_layer.py:154
↓ 2 callersMethodpreprocess
( self, input: torch.Tensor, exp_indices: torch.Tensor, full_scatter_indices:
python/triton_dist/layers/amd/ep_a2a_layer.py:277
↓ 2 callersFunctionpretty_triton_config_repr
()
python/triton_dist/tune.py:67
↓ 2 callersFunctionprint_benchmark_comparison
Print complete benchmark comparison for all configurations. Args: all_implementations: Dict mapping config keys to implementatio
python/triton_dist/profiler_utils.py:400
↓ 2 callersFunctionprint_bw_matrix
(title, matrix)
python/triton_dist/test/amd/test_bandwidth.py:575
↓ 2 callersFunctionprint_bw_matrix
(title: str, matrix: torch.Tensor, world_size: int)
python/triton_dist/test/amd/test_mori_shmem_bw.py:179
↓ 2 callersFunctionpromoteToShared
lib/Conversion/TritonDistributedToTritonGPU/TritonDistributedToTritonGPU.cpp:638
↓ 2 callersFunctionprune_fn_by_quatization
(config, A: torch.Tensor, B: torch.Tensor, *args, MAX_BLOCK_SIZE_M=256, MAX_BLOCK_SIZE_N=256,
python/triton_dist/kernels/amd/gemm.py:618
↓ 2 callersFunctionprune_fn_by_shared_mem
(config, A: torch.Tensor, B: torch.Tensor, *args, **kwargs)
python/triton_dist/kernels/amd/gemm.py:604
↓ 2 callersMethodpush_scope
Create a new nested scope (e.g., for function bodies, loops)
python/little_kernel/core/passes/utils/type_inference/infer_type.py:79
↓ 2 callersFunctionput
(ctx, ts, rank, dst_rank, num_sm)
python/triton_dist/test/nvidia/test_pp.py:101
↓ 2 callersMethodqkv_pack_a2a
( self, qkv, seq_lens_cpu: Optional[torch.Tensor] = None, num_comm_sms: int =
python/triton_dist/kernels/nvidia/sp_ulysess_qkv_gemm_all2all.py:889
↓ 2 callersFunctionquant_bf16_fp8
(tensor: torch.Tensor, gsize: int = 128)
python/triton_dist/test/nvidia/ep_a2a_utils.py:76
↓ 2 callersMethodrand_fill_kv_cache
(self, offset: int)
python/triton_dist/models/kv_cache.py:52
↓ 2 callersFunctionrand_tensor
(shape: list[int], dtype: torch.dtype)
python/triton_dist/test/nvidia/test_pp_block.py:98
↓ 2 callersFunctionread_file
Read file contents and return as list of lines.
scripts/diff_to_conflict.py:38
↓ 2 callersFunctionrecv
(ctx, rank, src_rank, num_sm)
python/triton_dist/test/nvidia/test_pp.py:120
↓ 2 callersFunctionreduce_scatter_ring_push_1d_intra_node_ce
( rank, num_ranks, input_tensor: torch.Tensor, input_flag: torch.Tensor, symm_reduce_tenso
python/triton_dist/kernels/nvidia/reduce_scatter.py:243
↓ 2 callersFunctionreduce_topk_non_tma_kernel
( input_ptr, # of shape (M * topk, N) stride (stride_m, stride_n) bias_ptr, # None, or of of shape (
python/triton_dist/kernels/nvidia/moe_utils.py:396
↓ 2 callersFunctionref_moe
out[t] = sum_j w[t,j] * relu(x[t] @ W1[e]) @ W2[e] (model-dtype matmul, fp32 reduce).
python/triton_dist/test/amd/test_ep_moe_fused.py:163
↓ 2 callersFunctionref_moe
out[t] = sum_j w[t,j] * relu(x[t] @ W1[e]) @ W2[e] (model-dtype matmul, fp32 reduce).
python/triton_dist/test/amd/test_ep_a2a_fused_kernel.py:118
↓ 2 callersFunctionref_paged_attn
( query: torch.Tensor, key_cache: torch.Tensor, value_cache: torch.Tensor, query_lens: List[in
python/triton_dist/mega_triton_kernel/test/torch_impl_utils.py:57
↓ 2 callersFunctionreference_expert_counts
Global bincount of expert indices (including drop token at index num_experts).
python/little_kernel/design/test_flashcomm_compute.py:80
↓ 2 callersFunctionreference_global_rank
Stable within-expert offset: out[p] = #{q < p | flat_idx[q] == flat_idx[p]}
python/little_kernel/design/test_flashcomm_compute.py:60
↓ 2 callersMethodregister_bool_op
Register a boolean operator handler.
python/little_kernel/core/passes/utils/registries/operator_registry.py:67
← previousnext →801–900 of 5,103, ranked by callers