MCPcopy Create free account

hub / github.com/bytedance/flux / functions

Functions784 in github.com/bytedance/flux

Functionatomic_store_sys
include/flux/cuda/cuda_common_device.hpp:93
Methodauto_comm_spec
include/flux/gemm_hparams.h:289
Methodauto_gemm_kind
include/flux/gemm_hparams.h:325
Methodauto_mainloop_stage
include/flux/gemm_hparams.h:335
Methodauto_raster_order
include/flux/gemm_hparams.h:353
Methodauto_tile_shape
include/flux/gemm_hparams.h:295
Functionbarrier_all
pynvshmem/src/functions.cpp:68
Functionbarrier_on_stream
pynvshmem/src/functions.cpp:74
Methodbegin_step
src/reduce_scatter/epilogue_evt.hpp:234
Methodbegin_step
src/reduce_scatter/epilogue_evt_nvshmem.hpp:224
Methodbegin_step
src/reduce_scatter/gemmk_visitor_load.hpp:127
Functionbitwise_check
src/ths_op/helper_ops.cc:26
Functionbootstrap_c10_pg_allgather
pynvshmem/bootstrap/torch/bootstrap_torch.cpp:49
Functionbootstrap_c10_pg_alltoall
pynvshmem/bootstrap/torch/bootstrap_torch.cpp:78
Functionbootstrap_c10_pg_barrier
pynvshmem/bootstrap/torch/bootstrap_torch.cpp:36
Functionbootstrap_c10_pg_finalize
pynvshmem/bootstrap/torch/bootstrap_torch.cpp:114
Functionbootstrap_c10_pg_global_exit
pynvshmem/bootstrap/torch/bootstrap_torch.cpp:108
Methodbuffer_ptr
src/reduce_scatter/reduce_scatter_kernel.hpp:479
Methodbuffer_tile
src/reduce_scatter/reduce_scatter_kernel.hpp:336
Methodcacheline_align_up
Pad the given allocation size up to the nearest cache line
src/all_gather/sm80_all_gather_gemm.hpp:331
Methodcacheline_align_up
Pad the given allocation size up to the nearest cache line
src/reduce_scatter/epilogue_evt.hpp:169
Methodcacheline_align_up
Pad the given allocation size up to the nearest cache line
src/reduce_scatter/epilogue_evt_nvshmem.hpp:145
Methodcacheline_align_up
Pad the given allocation size up to the nearest cache line
src/reduce_scatter/gemmk_universal_with_visitor_streamk.h:278
Methodcan_implement
Determines whether the GEMM problem size satisfies this kernel's alignment requirements
src/all_gather/sm80_all_gather_gemm.hpp:560
Methodcan_implement
src/all_gather/sm90_all_gather_gemm_tma_warpspecialized_cooperative.hpp:291
Methodcan_implement
src/reduce_scatter/epilogue_reduce_scatter.hpp:136
Methodcan_implement
src/reduce_scatter/epilogue_vectorized_reduce_scatter.hpp:146
Methodcan_implement
src/reduce_scatter/epilogue_nvshmem_reduce_scatter.hpp:158
Methodcan_implement
Determines whether the GEMM problem size satisfies this kernel's alignment requirements
src/reduce_scatter/gemmk_universal_with_visitor_streamk.h:509
Methodcan_implement
src/reduce_scatter/sm90_gemm_tma_warpspecialized_cooperative_reduce_scatter.hpp:239
Methodcheck_length
include/flux/flux.h:960
Methodcheck_type
include/flux/gemm_meta.h:166
Methodcheck_type
include/flux/gemm_meta.h:252
Methodcheck_value
Uses thread[0] to wait for at least the specified count of signals on the given flag counter
include/flux/cuda/system_barrier.hpp:57
FunctioncloseIpcMemHandleCallback
src/ths_op/ths_op.cc:84
Methodconstexpr
include/flux/cuda/cutlass_v3_builder.hpp:345
Methodcoord
src/reduce_scatter/reduce_scatter_kernel.hpp:453
Methodcopy
include/flux/cuda/memory_utils.hpp:259
Methodcopy
include/flux/cuda/memory_utils.hpp:282
Methodcopy
include/flux/cuda/memory_utils.hpp:306
Methodcopy
include/flux/cuda/memory_utils.hpp:337
Methodcopy_2d_ring
src/all_gather/ths_op/all_gather_gemm_kernel_crossnode.cc:565
Methodcopy_all2all
src/all_gather/ths_op/all_gather_gemm_kernel_crossnode.cc:422
Methodcopy_all_to_all
src/all_gather/ths_op/all_gather_gemm_kernel.cc:817
Methodcopy_local
src/all_gather/ths_op/all_gather_gemm_kernel_crossnode.cc:393
Methodcopy_local_ring
src/all_gather/ths_op/all_gather_gemm_kernel_crossnode.cc:672
Methodcopy_remote
src/reduce_scatter/epilogue_evt_nvshmem.hpp:300
Methodcopy_ring_1d_push
src/all_gather/ths_op/all_gather_gemm_kernel_crossnode.cc:734
Methodcopy_ring_pull
src/all_gather/ths_op/all_gather_gemm_kernel.cc:837
Methodcopy_ring_push_1d
src/all_gather/ths_op/all_gather_gemm_kernel.cc:863
Methodcopy_ring_push_2d_pcie
src/all_gather/ths_op/all_gather_gemm_kernel.cc:885
Methodcopy_to_next_node
src/all_gather/ths_op/all_gather_gemm_kernel_crossnode.cc:539
Methodcopy_unpack
include/flux/cuda/memory_utils.hpp:411
Functioncudaipc_barrier_all_on_stream
src/ths_op/helper_ops.cc:45
Methoddata_tile
src/reduce_scatter/reduce_scatter_kernel.hpp:408
Methoddebug_print_v
src/reduce_scatter/sm90_epilogue_evt.hpp:238
Methoddebug_print_v
src/reduce_scatter/sm90_reduce_scatter_utils.hpp:217
Methoddefault_gemm_kernel
include/flux/cuda/gemm_impls/gemm_grouped_impl.hpp:108
Methoddefault_tile_scheduler
include/flux/cuda/gemm_impls/gemm_v3_impl.hpp:56
Methodderived
include/flux/cuda/gemm_impls/gemm_operator_base_default_impl.hpp:46
Methoddo_epilogue
Perform epilogue computations and output
src/all_gather/sm80_all_gather_gemm.hpp:739
Methoddo_epilogue
Perform epilogue computations and output
src/reduce_scatter/gemmk_universal_with_visitor_streamk.h:712
Methodelapsed_millis
Return the elapsed time (in milliseconds)
include/flux/cuda/cuda_common.h:195
Methodend_epilogue
/
src/reduce_scatter/epilogue_evt.hpp:328
Methodend_epilogue
src/reduce_scatter/epilogue_evt_nvshmem.hpp:306
Methodend_step
src/reduce_scatter/epilogue_evt.hpp:257
Methodend_step
src/reduce_scatter/epilogue_evt_nvshmem.hpp:264
Methodenumerate_split_meta_hparams_pairs
src/comm_none/gemm_v2_comm_none.hpp:210
Methodenumerate_split_meta_hparams_pairs
include/flux/gemm_operator_base.h:71
Methodeq
include/flux/flux.h:905
Functionestimate_gemm_sol_time_ms
refer to this: https://arnon.dk/matching-sm-architectures-arch-and-gencode-for-various-nvidia-cards/
python/flux/util.py:44
Functionexec_in_rank_order
(group: torch.distributed.ProcessGroup, func: Callable)
python/flux/dist_utils.py:79
Methodfetch_next_work
src/all_gather/sm90_all_gather_gemm_tma_warpspecialized_cooperative.hpp:773
Methodfetch_next_work
src/reduce_scatter/sm90_gemm_tma_warpspecialized_cooperative_reduce_scatter.hpp:735
Methodfields
include/flux/flux.h:955
Methodfilter_fast_accum
include/flux/gemm_meta.h:386
Functionfilter_kernel_schedule
include/flux/gemm_hparams.h:433
Methodfilter_layout
include/flux/gemm_meta.h:401
Methodflags
src/reduce_scatter/reduce_scatter_kernel.hpp:310
Functionflux_create_tensor
src/ths_op/flux_shm.cc:170
Methodforward
( self, input: torch.Tensor, weight: torch.Tensor, bias: torch.Tensor )
python/flux/gemm_rs_sm80.py:135
Methodforward
src/all_gather/ths_op/all_gather_gemm_kernel_crossnode.cc:376
Methodforward
src/reduce_scatter/ths_op/gemm_reduce_scatter.cc:628
Methodforward_allgather
src/all_gather/ths_op/all_gather_gemm_kernel.cc:529
Methodforward_barrier
src/reduce_scatter/ths_op/gemm_reduce_scatter.cc:607
Methodforward_gemm
src/all_gather/ths_op/all_gather_gemm_kernel.cc:592
Methodforward_gemm
src/reduce_scatter/ths_op/gemm_reduce_scatter.cc:587
Methodforward_gemm_impl
src/all_gather/ths_op/all_gather_gemm_kernel.cc:440
Methodforward_gemm_impl
src/reduce_scatter/ths_op/gemm_reduce_scatter.cc:345
Methodforward_impl
src/all_gather/ths_op/all_gather_gemm_kernel_crossnode.cc:299
Methodforward_reduce_scatter
src/reduce_scatter/ths_op/gemm_reduce_scatter.cc:621
Methodforward_reduce_scatter_impl
src/reduce_scatter/ths_op/gemm_reduce_scatter.cc:481
Methodfrom_tuple
include/flux/flux.h:966
Methodfrom_tuple
include/flux/cuda/cutlass_v3_builder.hpp:96
Methodgemm
Executes one GEMM
src/all_gather/sm80_all_gather_gemm.hpp:848
Methodgemm
Executes one GEMM
src/reduce_scatter/gemmk_universal_with_visitor_streamk.h:829
Methodgemm_device
///////////////////// CRTP functions /////////////////////
include/flux/cuda/gemm_impls/gemm_grouped_impl.hpp:138
Methodgemm_device
///////////////////// CRTP functions /////////////////////
include/flux/cuda/gemm_impls/gemm_v2_impl.hpp:348
Methodgemm_kernel
src/comm_none/gemm_v2_comm_none.hpp:48
Methodgemm_kernel
src/all_gather/gemm_v2_ag_kernel.hpp:64
← previousnext →401–500 of 784, ranked by callers