MCPcopy Create free account

hub / github.com/bytedance/flux / functions

Functions784 in github.com/bytedance/flux

↓ 1 callersFunctiongen_tuning_space
(check: bool)
tools/tune_gemm_rs.py:67
↓ 1 callersMethodget
include/flux/op_registry.h:107
↓ 1 callersFunctionget_device_path
src/ths_op/topo_utils.cc:94
↓ 1 callersMethodget_gemm_meta
src/comm_none/ths_op/gemm_only.cc:54
↓ 1 callersMethodget_gemm_meta
src/all_gather/ths_op/all_gather_gemm_kernel.cc:258
↓ 1 callersMethodget_gemm_meta
src/reduce_scatter/ths_op/gemm_reduce_scatter.cc:289
↓ 1 callersFunctionget_inter_node_ag_gemm_group
(tp_group: dist.ProcessGroup, nnodes: int)
python/flux/ag_gemm.py:26
↓ 1 callersFunctionget_inter_node_ag_group
(tp_group: dist.ProcessGroup, nnodes: int)
python/flux/ag_kernel_crossnode.py:27
↓ 1 callersFunctionget_intra_node_ag_gemm_group
(tp_group: dist.ProcessGroup, nnodes: int)
python/flux/ag_gemm.py:48
↓ 1 callersFunctionget_intra_node_ag_group
(tp_group: dist.ProcessGroup, nnodes: int)
python/flux/ag_kernel_crossnode.py:50
↓ 1 callersFunctionget_local_version
(public_version)
setup.py:60
↓ 1 callersFunctionget_numa_id
src/ths_op/topo_utils.cc:118
↓ 1 callersFunctionget_package_version
()
setup.py:73
↓ 1 callersFunctionget_pcie_gen
src/cuda/cuda_common.cc:51
↓ 1 callersFunctionget_profiler
(exp_name, warmups=1, iters=1)
test/utils.py:85
↓ 1 callersFunctionget_rank_from_env_impl
src/cuda/utils.cc:74
↓ 1 callersMethodget_rt_conf
src/comm_none/ths_op/gemm_only.cc:77
↓ 1 callersMethodget_rt_conf
src/reduce_scatter/ths_op/gemm_reduce_scatter.cc:316
↓ 1 callersMethodget_tma_tensor
include/flux/cuda/memory_utils.hpp:395
↓ 1 callersFunctionget_torch_output
(input: torch.Tensor, weight: torch.Tensor, bias: torch.Tensor)
tools/tune_ag_gemm_kernel.py:99
↓ 1 callersFunctionget_torch_output
(input: torch.Tensor, weight: torch.Tensor, bias: torch.Tensor)
tools/tune_gemm_rs.py:102
↓ 1 callersFunctionget_wheel_url
()
setup.py:190
↓ 1 callersFunctionget_world_size_from_env_impl
src/cuda/utils.cc:66
↓ 1 callersFunctionhas_nvlink_support
src/ths_op/topo_utils.cc:45
↓ 1 callersFunctioninit_dist_env_tp
src/ths_op/ths_op.cc:351
↓ 1 callersFunctioninit_dist_env_tp_with_ep
src/ths_op/ths_op.cc:363
↓ 1 callersMethodinit_global_group
(self)
python/flux/dist_utils.py:53
↓ 1 callersFunctioninit_moe_arguments
src/ths_op/ths_op.cc:378
↓ 1 callersMethodinit_output_buffer
src/reduce_scatter/ths_op/gemm_reduce_scatter.cc:102
↓ 1 callersFunctioninit_peer_access
src/reduce_scatter/test/test_gemm_rs.cc:30
↓ 1 callersFunctioninit_profiling_context
src/ths_op/ths_op.cc:335
↓ 1 callersFunctioninit_seed
()
test/utils.py:46
↓ 1 callersFunctioninit_topo
src/ths_op/topo_utils.cc:137
↓ 1 callersFunctioninit_tuning_record
src/ths_op/ths_op.cc:346
↓ 1 callersMethodinitialize_all
src/ths_op/ths_op.cc:269
↓ 1 callersMethodinitialize_args_workspace
include/flux/gemm_operator_base.h:97
↓ 1 callersMethodis_cuda_function
(self)
pynvshmem/tools/autogen_pybind.py:157
↓ 1 callersFunctionis_local_tp_group_initialized
()
test/utils.py:61
↓ 1 callersMethodis_source_needed
src/reduce_scatter/epilogue_reduce_scatter.hpp:145
↓ 1 callersMethodis_source_needed
src/reduce_scatter/epilogue_vectorized_reduce_scatter.hpp:156
↓ 1 callersMethodis_source_needed
src/reduce_scatter/epilogue_nvshmem_reduce_scatter.hpp:168
↓ 1 callersMethodkernel_params
src/all_gather/gemm_v2_ag_kernel.hpp:54
↓ 1 callersFunctionload_from_file
src/cuda/op_registry_proto_utils.cc:48
↓ 1 callersFunctionload_proto_from_file
src/cuda/op_registry_proto_utils.cc:37
↓ 1 callersFunctionmain
()
setup.py:237
↓ 1 callersMethodname
include/flux/flux.h:940
↓ 1 callersFunctionnccl_deps
()
setup.py:144
↓ 1 callersFunctionnvshmem_create_tensor
namespace
src/ths_op/flux_shm.cc:48
↓ 1 callersFunctionnvshmem_deps
()
setup.py:117
↓ 1 callersFunctionparse_args
()
tools/tune_ag_gemm_kernel.py:188
↓ 1 callersFunctionparse_args
()
tools/tune_gemm_rs.py:179
↓ 1 callersFunctionparse_args
()
test/test_gemm_only.py:136
↓ 1 callersFunctionparse_args
()
test/test_ag_kernel_functional.py:69
↓ 1 callersFunctionparse_args
()
test/test_ag_kernel_pyshmem.py:243
↓ 1 callersFunctionparse_args
()
test/test_gemm_rs.py:282
↓ 1 callersFunctionparse_args
()
test/test_ag_kernel.py:345
↓ 1 callersFunctionparse_args
()
test/test_ag_kernel_crossnode.py:273
↓ 1 callersFunctionparse_args
()
pynvshmem/tools/autogen_pybind.py:398
↓ 1 callersFunctionparse_args
()
pynvshmem/examples/run_example.py:55
↓ 1 callersFunctionperf_flux
( input: torch.Tensor, weight: torch.Tensor, bias: torch.Tensor, input_scale: torch.Tensor,
test/test_gemm_only.py:92
↓ 1 callersFunctionperf_flux
( input: torch.Tensor, weight: torch.Tensor, bias: torch.Tensor, warmup: int, iters: int,
test/test_ag_kernel_pyshmem.py:139
↓ 1 callersFunctionperf_flux
( input: torch.Tensor, weight: torch.Tensor, bias: torch.Tensor, transpose_weight: bool, f
test/test_gemm_rs.py:173
↓ 1 callersFunctionperf_flux
( input: torch.Tensor, weight: torch.Tensor, bias: torch.Tensor, input_scale: torch.Tensor,
test/test_ag_kernel.py:172
↓ 1 callersFunctionperf_flux
( input: torch.Tensor, weight: torch.Tensor, warmup: int, iters: int, transpose_weight: bo
test/test_ag_kernel_crossnode.py:145
↓ 1 callersFunctionperf_torch
( input: torch.Tensor, weight: torch.Tensor, bias: torch.Tensor, input_scale: torch.Tensor,
test/test_gemm_only.py:65
↓ 1 callersFunctionperf_torch
( input: torch.Tensor, weight: torch.Tensor, bias: torch.Tensor, warmup: int, iters: int )
test/test_ag_kernel_pyshmem.py:82
↓ 1 callersFunctionperf_torch
( input: torch.Tensor, weight: torch.Tensor, bias: torch.Tensor, warmup: int, iters: int,
test/test_gemm_rs.py:78
↓ 1 callersFunctionperf_torch
( input: torch.Tensor, weight: torch.Tensor, bias: torch.Tensor, input_scale: torch.Tensor,
test/test_ag_kernel.py:96
↓ 1 callersFunctionperf_torch
(input: torch.Tensor, weight: torch.Tensor, warmup: int, iters: int)
test/test_ag_kernel_crossnode.py:92
↓ 1 callersMethodprofiling
src/all_gather/ths_op/all_gather_gemm_kernel.cc:712
↓ 1 callersMethodprofiling
src/reduce_scatter/ths_op/gemm_reduce_scatter.cc:670
↓ 1 callersFunctionread_flux_ths_files
()
setup.py:102
↓ 1 callersMethodrun
(self)
setup.py:211
↓ 1 callersFunctionrun_flux_profiling
( prof_ctx: flux.ProfilingContext, input: torch.Tensor, weight: torch.Tensor, bias: torch.Tens
tools/tune_ag_gemm_kernel.py:118
↓ 1 callersFunctionrun_flux_profiling
( prof_ctx: flux.ProfilingContext, input: torch.Tensor, weight: torch.Tensor, bias: torch.Tens
tools/tune_gemm_rs.py:119
↓ 1 callersFunctionrun_flux_with_op
( ag_op: flux.AGKernel, input: torch.Tensor, weight: torch.Tensor, bias: torch.Tensor, loc
test/test_ag_kernel_functional.py:54
↓ 1 callersFunctionrun_gemm
src/all_gather/ths_op/all_gather_gemm_kernel_crossnode.cc:87
↓ 1 callersFunctionrun_memcpy
src/all_gather/ths_op/all_gather_gemm_kernel_crossnode.cc:97
↓ 1 callersFunctionrun_rs
src/reduce_scatter/test/test_gemm_rs.cc:42
↓ 1 callersFunctionrun_test_with_args
( M_max, local_N, K, dtype, transpose_weight: bool, local_copy: bool, has_bias: bo
test/test_ag_kernel_functional.py:83
↓ 1 callersFunctionrun_torch
( input: torch.Tensor, weight: torch.Tensor, bias: torch.Tensor, )
test/test_ag_kernel_functional.py:33
↓ 1 callersMethodsetup_deterministic
(self, init_seed: int)
python/flux/dist_utils.py:38
↓ 1 callersFunctionsetup_pytorch_extension
Setup CppExtension for PyTorch support
setup.py:152
↓ 1 callersMethodsm80_collective_epilogue
src/reduce_scatter/gemm_v3_reduce_scatter.hpp:85
↓ 1 callersMethodsm90_collective_epilogue
src/reduce_scatter/gemm_v3_reduce_scatter.hpp:112
↓ 1 callersMethodstream_operator_fields_impl
include/flux/flux.h:982
↓ 1 callersMethodswizzle_m
src/reduce_scatter/tile_scheduler/tile_mappings.hpp:113
↓ 1 callersMethodswizzle_rank
src/reduce_scatter/tile_scheduler/tile_mappings.hpp:58
↓ 1 callersMethodtb_swizzle
src/reduce_scatter/gemm_v2_reduce_scatter.hpp:182
↓ 1 callersMethodtile_finished
src/all_gather/sm80_all_gather_gemm.hpp:520
↓ 1 callersMethodtile_finished
src/reduce_scatter/gemmk_universal_with_visitor_streamk.h:469
↓ 1 callersMethodtile_scheduler
src/comm_none/gemm_v3_comm_none.hpp:49
↓ 1 callersMethodto_make_expr_fields_impl
include/flux/flux.h:990
↓ 1 callersFunctionto_make_expression_tuple_impl
include/flux/flux.h:872
↓ 1 callersFunctionto_std_tuple_impl
include/flux/flux.h:562
↓ 1 callersFunctiontune_one_config
(prof_ctx: flux.ProfilingContext, config: TuningConfig)
tools/tune_ag_gemm_kernel.py:151
↓ 1 callersFunctiontune_one_config
(prof_ctx: flux.ProfilingContext, config: TuningConfig)
tools/tune_gemm_rs.py:149
↓ 1 callersFunctiontuple_enumerate_impl
include/flux/flux.h:397
↓ 1 callersFunctiontuple_filter_impl
include/flux/flux.h:261
↓ 1 callersFunctiontuple_filter_impl_element
include/flux/flux.h:249
← previousnext →201–300 of 784, ranked by callers