MCPcopy Create free account

hub / github.com/ByteDance-Seed/Triton-distributed / functions

Functions5,103 in github.com/ByteDance-Seed/Triton-distributed

↓ 1 callersFunction_has_fullmesh_xgmi_amdsmi
()
python/triton_dist/amd_utils.py:173
↓ 1 callersFunction_has_fullmesh_xgmi_rocm
()
python/triton_dist/amd_utils.py:187
↓ 1 callersFunction_has_slice_intersection_for_diff_shape
(start_indices1, data_sizes1, shape1, start_indices2, data_sizes2, shape2)
python/triton_dist/mega_triton_kernel/core/utils.py:37
↓ 1 callersFunction_indexed_signature
(src: triton.compiler.ASTSource)
python/triton_dist/tools/compile/compile.py:156
↓ 1 callersMethod_infer_builtin_call
Infer type for builtin functions with eval_return_type.
python/little_kernel/core/passes/utils/type_inference/type_inference_call.py:166
↓ 1 callersMethod_infer_cpp_type
Infer C++ type from Python value node.
python/little_kernel/codegen/special_struct/struct_converter.py:231
↓ 1 callersMethod_infer_kernel_call
Infer type for LLKernel calls.
python/little_kernel/core/passes/utils/type_inference/type_inference_call.py:235
↓ 1 callersMethod_infer_method_call
Infer type for method calls (e.g., obj.method()).
python/little_kernel/core/passes/utils/type_inference/type_inference_call.py:95
↓ 1 callersMethod_infer_return_expr_type
Infer type of a return expression node.
python/little_kernel/core/passes/utils/type_inference/method_resolver.py:81
↓ 1 callersMethod_infer_special_struct_attribute
Infer type of a special struct attribute by analyzing __init__ method.
python/little_kernel/core/passes/utils/type_inference/type_inference_ast_visitors.py:352
↓ 1 callersMethod_infer_special_struct_constructor
Infer type for special struct constructors.
python/little_kernel/core/passes/utils/type_inference/type_inference_call.py:246
↓ 1 callersMethod_infer_unresolved_call
Handle unresolved function calls (method calls on variables).
python/little_kernel/core/passes/utils/type_inference/type_inference_call.py:267
↓ 1 callersMethod_init_context
Initialize CUDA context.
python/little_kernel/runtime/cuda_runtime.py:298
↓ 1 callersMethod_init_ctx
(self)
python/triton_dist/layers/nvidia/gemm_allreduce_layer.py:79
↓ 1 callersMethod_init_cuda_graph
(self, bsz: int = 1)
python/triton_dist/models/engine.py:75
↓ 1 callersMethod_init_kv_cache
(self, bsz: int)
python/triton_dist/models/engine.py:61
↓ 1 callersMethod_init_model
(self)
python/triton_dist/models/engine.py:55
↓ 1 callersFunction_install_triton_dist_hook
()
python/triton_dist/jit.py:268
↓ 1 callersMethod_is_const_method
Check if method should be const (doesn't modify member variables).
python/little_kernel/codegen/special_struct/struct_converter.py:796
↓ 1 callersMethod_is_constexpr_condition
Check if condition should use constexpr if (template parameters, enum comparisons, or ll.const).
python/little_kernel/codegen/special_struct/translator.py:310
↓ 1 callersFunction_is_cuda_launch_blocking
()
python/triton_dist/utils.py:924
↓ 1 callersFunction_is_gpu_master
()
python/triton_dist/kernels/amd/common_ops.py:51
↓ 1 callersFunction_is_little_kernel_related
True if value is safe to add to ctx (whitelist for closure vars).
python/little_kernel/core/compile.py:34
↓ 1 callersFunction_is_local_interface
(interface)
python/triton_dist/kernels/nvidia/comm_perf_model.py:39
↓ 1 callersMethod_is_local_variable
Check if a variable name is defined locally (should not be resolved from ctx). Supports both: - scope_vars (Dict[str
python/little_kernel/codegen/visitors/expression_codegen.py:53
↓ 1 callersFunction_is_one_shot
(method: AllReduceMethod)
python/triton_dist/test/nvidia/test_allreduce.py:146
↓ 1 callersMethod_is_parameter_modified
Check if a parameter is modified in the method body.
python/little_kernel/codegen/special_struct/struct_converter.py:809
↓ 1 callersFunction_is_valid_arg_sig
(sig: str)
python/triton_dist/tools/compile_aot.py:160
↓ 1 callersFunction_list_to_intervals
(nums)
python/triton_dist/mega_triton_kernel/core/graph.py:33
↓ 1 callersFunction_load
(filename)
python/triton_dist/kernels/nvidia/threadblock_swizzle_ag_moe_triton.py:433
↓ 1 callersMethod_load_library
Load libcuda.so library with robust path detection. Tries multiple methods to find libcuda: 1. Environment variable LITTLE_KE
python/little_kernel/runtime/cuda_runtime.py:71
↓ 1 callersMethod_load_library
Load libcuda.so library.
python/little_kernel/runtime/tma_descriptor.py:93
↓ 1 callersFunction_load_v2_impl
(ptr, suffix: core.constexpr, _semantic=None)
python/triton_dist/language/extra/cuda/language_extra.py:166
↓ 1 callersFunction_make_const_sig
(src: triton.compiler.ASTSource)
python/triton_dist/tools/compile/compile.py:165
↓ 1 callersFunction_make_data
(M)
tutorials/08-overlapping-gemm-reduce-scatter.py:482
↓ 1 callersFunction_make_tensor
rand() * scale + bias randint(-scale, scale) + bias
python/triton_dist/utils.py:410
↓ 1 callersFunction_make_triton_algo_info_with_schema
(algo_info: str, schema: List[Tuple[str, type]])
python/triton_dist/tools/compile_aot.py:300
↓ 1 callersMethod_map_arguments
Map call arguments (positional/keyword) to function parameters
python/little_kernel/core/passes/inline.py:184
↓ 1 callersFunction_match_col
(filename)
python/triton_dist/tools/tune/find_topk.py:128
↓ 1 callersFunction_materialize_constexpr
( signature: str, grid: List[str], kernel: triton.JITFunction, triton_algo_inf
python/triton_dist/tools/compile_aot.py:256
↓ 1 callersMethod_merge_all_trace
(self, trace_content_list)
python/triton_dist/profiler_utils.py:262
↓ 1 callersFunction_merge_json
( to_merge_files: List[Path], output_json: Path, compress: bool = True, version: int = 2, )
python/triton_dist/profiler_utils.py:193
↓ 1 callersFunction_merge_json_v1
(to_merge_files: List[Path], output_json: Path, compress: bool = True)
python/triton_dist/profiler_utils.py:100
↓ 1 callersFunction_merge_json_v2
( to_merge_files: List[Path], output_json: Path, compress: bool = True, )
python/triton_dist/profiler_utils.py:170
↓ 1 callersFunction_meta_sig
(num_stages: int, num_warps: int)
python/triton_dist/tools/compile/compile.py:54
↓ 1 callersFunction_parse_args
()
python/triton_dist/tools/compile_aot.py:774
↓ 1 callersFunction_parse_args
()
python/triton_dist/test/nvidia/test_allreduce.py:184
↓ 1 callersFunction_parse_args
()
python/triton_dist/kernels/nvidia/threadblock_swizzle_ag_moe_triton.py:419
↓ 1 callersFunction_parse_nvml_field_value
(fv)
python/triton_dist/nv_utils.py:128
↓ 1 callersFunction_parse_rocm_shownodesbw_output_in_gbps
$ rocm-smi --shownodesbw ============================ ROCm System Management Interface ============================ ===============
python/triton_dist/amd_utils.py:307
↓ 1 callersMethod_pp_get
Triton Dist PP get operation based on test_pp.py
python/triton_dist/layers/nvidia/pp_block.py:172
↓ 1 callersMethod_pp_put
Triton Dist PP put operation based on test_pp.py
python/triton_dist/layers/nvidia/pp_block.py:160
↓ 1 callersFunction_pretty_format
(nbytes)
python/triton_dist/test/nvidia/test_allreduce.py:61
↓ 1 callersFunction_process_item
(item, rank, delta)
python/triton_dist/profiler_utils.py:78
↓ 1 callersFunction_python_value_to_ast_node
Convert a Python default value to an AST node. Handles: - Basic types (int, float, bool, str, None) -> ast.Constant - Enum insta
python/little_kernel/core/passes/special_struct_materialize_pass.py:44
↓ 1 callersFunction_qkv_pack_attn_fwd
( tile_id, qkv_ptr, out_ptr, # N_CTX, # H_Q: tl.constexpr, H_KV: tl.constexpr, S
python/triton_dist/mega_triton_kernel/kernels/flash_attn.py:91
↓ 1 callersFunction_qkv_pack_qk_norm_rope_split_v_kernel
qkv: (bs, seq_len, num_total_heads, head_dim) BLOCK_HD equal to next_power_of_2(head_dim)
python/triton_dist/mega_triton_kernel/kernels/norm.py:238
↓ 1 callersFunction_rand
()
python/triton_dist/kernels/nvidia/threadblock_swizzle_ag_moe_triton.py:335
↓ 1 callersFunction_rand
()
python/triton_dist/kernels/nvidia/threadblock_swizzle_ag_moe.py:196
↓ 1 callersFunction_randint_with_align
(max_M, alignment: int)
python/triton_dist/test/nvidia/test_allreduce.py:84
↓ 1 callersFunction_random_straggler_option
()
python/triton_dist/test/nvidia/test_allreduce.py:88
↓ 1 callersFunction_recv_ll_and_multimem_st_ll_block
split src/dest outside of _recv_ll. this function is designed for a threadblock num_ints: of the pre-LL-packed num_ints.
python/triton_dist/kernels/nvidia/low_latency_allgather.py:589
↓ 1 callersMethod_register_default_handlers
Register default codegen handlers for loop modifiers.
python/little_kernel/codegen/registries/loop_modifier_codegen.py:39
↓ 1 callersMethod_register_default_modifiers
Register default loop modifiers.
python/little_kernel/core/passes/utils/registries/loop_modifier_registry.py:74
↓ 1 callersMethod_register_default_operators
Register default operators that return the left operand type.
python/little_kernel/core/passes/utils/registries/operator_registry.py:40
↓ 1 callersFunction_remote_ptr_wrapper
(local_ptr, pe, _semantic=None)
python/triton_dist/language/extra/cuda/libnvshmem_device.py:176
↓ 1 callersMethod_replace_empty_assigns_in_body
Replace ll.empty assigns with alloc/slice in body and nested blocks. No hardcoded sizes.
python/little_kernel/core/passes/insert_mem_alloc.py:176
↓ 1 callersMethod_replace_returns
Replace return statements with assignments to the LHS variable (for functions with return values)
python/little_kernel/core/passes/inline.py:251
↓ 1 callersMethod_resolve_conflicts
Rename local variables in inlined function to avoid conflicts with: - Variables in the calling context - Parameters of the ca
python/little_kernel/core/passes/inline.py:208
↓ 1 callersMethod_resolve_function
Resolve the function being called.
python/little_kernel/core/passes/utils/type_inference/type_inference_call.py:130
↓ 1 callersMethod_resolve_subscript_annotation
Resolve subscript annotations like template[...], Tuple[...], or const[...]
python/little_kernel/core/passes/utils/type_inference/type_inference_visitors.py:110
↓ 1 callersFunction_resolve_template_param
Generic function to resolve a template parameter from AST node to C++ constant expression. This is a generic version that works for any
python/little_kernel/codegen/special_struct/utils.py:38
↓ 1 callersFunction_run_all_gather_triton
()
tutorials/03a-inter-node-allgather.py:127
↓ 1 callersFunction_run_all_gather_triton
()
tutorials/03-inter-node-allgather.py:212
↓ 1 callersFunction_run_with_ag_op
()
python/triton_dist/test/nvidia/test_fast_allgather.py:69
↓ 1 callersMethod_setup_buffers
Initializes buffers for communication.
python/triton_dist/test/nvidia/test_pp_block.py:134
↓ 1 callersFunction_split_tiles_for_each_segment
python/triton_dist/kernels/nvidia/threadblock_swizzle_ag_moe.cc:109
↓ 1 callersFunction_split_tiles_for_each_segment
( expert_id, rank, tp_size, block_size_m, token_cnts: List[int], )
python/triton_dist/kernels/nvidia/threadblock_swizzle_ag_moe.py:55
↓ 1 callersMethod_step
Executes one fundamental step of the pipeline for the current stage.
python/triton_dist/test/nvidia/test_pp_block.py:148
↓ 1 callersFunction_store_v2_impl
(ptr, val0, val1, suffix: core.constexpr, _semantic=None)
python/triton_dist/language/extra/cuda/language_extra.py:225
↓ 1 callersFunction_str_to_dist_comm_scopre
(comm_scope)
python/triton_dist/language/distributed_ops.py:42
↓ 1 callersFunction_str_to_dist_signal_op
(sig_op)
python/triton_dist/language/distributed_ops.py:30
↓ 1 callersMethod_substitute_parameters
Replace function parameters with call arguments and rename conflicting variables
python/little_kernel/core/passes/inline.py:229
↓ 1 callersFunction_team_translate_pe
(src_team, pe_in_src_team, dest_team, _semantic=None)
python/triton_dist/language/extra/cuda/libnvshmem_device.py:933
↓ 1 callersFunction_test_atomic_add
(semantic, scope, dtype)
python/triton_dist/test/common/test_language_extra.py:86
↓ 1 callersFunction_test_atomic_cas
(semantic, scope, dtype)
python/triton_dist/test/common/test_language_extra.py:33
↓ 1 callersFunction_test_ld_st
(ld_semantic, st_semantic, scope, dtype)
python/triton_dist/test/common/test_language_extra.py:133
↓ 1 callersFunction_to_c_value
(x)
python/triton_dist/tools/compile_aot.py:399
↓ 1 callersFunction_to_rocm_scope
(scope: core.constexpr = core.constexpr("gpu"), _semantic=None)
python/triton_dist/kernels/common_ops.py:83
↓ 1 callersFunction_to_rocm_semantic
(semantic: core.constexpr = core.constexpr("sc"), _semantic=None)
python/triton_dist/kernels/common_ops.py:104
↓ 1 callersFunction_to_schema
(triton_algo_infos)
python/triton_dist/tools/compile_aot.py:253
↓ 1 callersFunction_to_value
(value, ctype)
python/triton_dist/tools/compile_aot.py:302
↓ 1 callersFunction_torch_dtype_from_str
(s: str)
python/triton_dist/tune.py:154
↓ 1 callersFunction_torch_func
()
python/triton_dist/benchmark/bench_allgather_gemm.py:101
↓ 1 callersFunction_torch_has_fp8
()
python/triton_dist/utils.py:987
↓ 1 callersFunction_torch_impl
(input, weight, seq_lens_cpu, input_scale, weight_scale, bias)
python/triton_dist/test/nvidia/test_llm_ulysess_gemm_all2all_intra_node.py:256
↓ 1 callersFunction_torch_impl
(input, weight, bias, seq_lens_cpu)
python/triton_dist/test/nvidia/test_llm_ulysess_all2all_gemm_intra_node.py:219
↓ 1 callersFunction_torch_impl
(input, seq_lens_cpu)
python/triton_dist/test/nvidia/test_llm_ulysess_post_attn_all2all_intra_node.py:189
↓ 1 callersFunction_torch_impl
(input, seq_lens_cpu)
python/triton_dist/test/nvidia/test_llm_ulysess_pre_attn_all2all_intra_node.py:261
↓ 1 callersFunction_track_iter
(profiler_buffer: np.ndarray, num_blocks, num_groups)
python/triton_dist/tools/profiler/viewer.py:54
← previousnext →1,101–1,200 of 5,103, ranked by callers