Code
Hub
Workspaces
Following
Trending
Connect
MCP
copy
Create free account
hub
/
github.com/bytedance/flux
/ functions
Functions
784 in github.com/bytedance/flux
⨍
Functions
784
◇
Types & classes
309
Function
atomic_store_sys
include/flux/cuda/cuda_common_device.hpp:93
Method
auto_comm_spec
include/flux/gemm_hparams.h:289
Method
auto_gemm_kind
include/flux/gemm_hparams.h:325
Method
auto_mainloop_stage
include/flux/gemm_hparams.h:335
Method
auto_raster_order
include/flux/gemm_hparams.h:353
Method
auto_tile_shape
include/flux/gemm_hparams.h:295
Function
barrier_all
pynvshmem/src/functions.cpp:68
Function
barrier_on_stream
pynvshmem/src/functions.cpp:74
Method
begin_step
src/reduce_scatter/epilogue_evt.hpp:234
Method
begin_step
src/reduce_scatter/epilogue_evt_nvshmem.hpp:224
Method
begin_step
src/reduce_scatter/gemmk_visitor_load.hpp:127
Function
bitwise_check
src/ths_op/helper_ops.cc:26
Function
bootstrap_c10_pg_allgather
pynvshmem/bootstrap/torch/bootstrap_torch.cpp:49
Function
bootstrap_c10_pg_alltoall
pynvshmem/bootstrap/torch/bootstrap_torch.cpp:78
Function
bootstrap_c10_pg_barrier
pynvshmem/bootstrap/torch/bootstrap_torch.cpp:36
Function
bootstrap_c10_pg_finalize
pynvshmem/bootstrap/torch/bootstrap_torch.cpp:114
Function
bootstrap_c10_pg_global_exit
pynvshmem/bootstrap/torch/bootstrap_torch.cpp:108
Method
buffer_ptr
src/reduce_scatter/reduce_scatter_kernel.hpp:479
Method
buffer_tile
src/reduce_scatter/reduce_scatter_kernel.hpp:336
Method
cacheline_align_up
Pad the given allocation size up to the nearest cache line
src/all_gather/sm80_all_gather_gemm.hpp:331
Method
cacheline_align_up
Pad the given allocation size up to the nearest cache line
src/reduce_scatter/epilogue_evt.hpp:169
Method
cacheline_align_up
Pad the given allocation size up to the nearest cache line
src/reduce_scatter/epilogue_evt_nvshmem.hpp:145
Method
cacheline_align_up
Pad the given allocation size up to the nearest cache line
src/reduce_scatter/gemmk_universal_with_visitor_streamk.h:278
Method
can_implement
Determines whether the GEMM problem size satisfies this kernel's alignment requirements
src/all_gather/sm80_all_gather_gemm.hpp:560
Method
can_implement
src/all_gather/sm90_all_gather_gemm_tma_warpspecialized_cooperative.hpp:291
Method
can_implement
src/reduce_scatter/epilogue_reduce_scatter.hpp:136
Method
can_implement
src/reduce_scatter/epilogue_vectorized_reduce_scatter.hpp:146
Method
can_implement
src/reduce_scatter/epilogue_nvshmem_reduce_scatter.hpp:158
Method
can_implement
Determines whether the GEMM problem size satisfies this kernel's alignment requirements
src/reduce_scatter/gemmk_universal_with_visitor_streamk.h:509
Method
can_implement
src/reduce_scatter/sm90_gemm_tma_warpspecialized_cooperative_reduce_scatter.hpp:239
Method
check_length
include/flux/flux.h:960
Method
check_type
include/flux/gemm_meta.h:166
Method
check_type
include/flux/gemm_meta.h:252
Method
check_value
Uses thread[0] to wait for at least the specified count of signals on the given flag counter
include/flux/cuda/system_barrier.hpp:57
Function
closeIpcMemHandleCallback
src/ths_op/ths_op.cc:84
Method
constexpr
include/flux/cuda/cutlass_v3_builder.hpp:345
Method
coord
src/reduce_scatter/reduce_scatter_kernel.hpp:453
Method
copy
include/flux/cuda/memory_utils.hpp:259
Method
copy
include/flux/cuda/memory_utils.hpp:282
Method
copy
include/flux/cuda/memory_utils.hpp:306
Method
copy
include/flux/cuda/memory_utils.hpp:337
Method
copy_2d_ring
src/all_gather/ths_op/all_gather_gemm_kernel_crossnode.cc:565
Method
copy_all2all
src/all_gather/ths_op/all_gather_gemm_kernel_crossnode.cc:422
Method
copy_all_to_all
src/all_gather/ths_op/all_gather_gemm_kernel.cc:817
Method
copy_local
src/all_gather/ths_op/all_gather_gemm_kernel_crossnode.cc:393
Method
copy_local_ring
src/all_gather/ths_op/all_gather_gemm_kernel_crossnode.cc:672
Method
copy_remote
src/reduce_scatter/epilogue_evt_nvshmem.hpp:300
Method
copy_ring_1d_push
src/all_gather/ths_op/all_gather_gemm_kernel_crossnode.cc:734
Method
copy_ring_pull
src/all_gather/ths_op/all_gather_gemm_kernel.cc:837
Method
copy_ring_push_1d
src/all_gather/ths_op/all_gather_gemm_kernel.cc:863
Method
copy_ring_push_2d_pcie
src/all_gather/ths_op/all_gather_gemm_kernel.cc:885
Method
copy_to_next_node
src/all_gather/ths_op/all_gather_gemm_kernel_crossnode.cc:539
Method
copy_unpack
include/flux/cuda/memory_utils.hpp:411
Function
cudaipc_barrier_all_on_stream
src/ths_op/helper_ops.cc:45
Method
data_tile
src/reduce_scatter/reduce_scatter_kernel.hpp:408
Method
debug_print_v
src/reduce_scatter/sm90_epilogue_evt.hpp:238
Method
debug_print_v
src/reduce_scatter/sm90_reduce_scatter_utils.hpp:217
Method
default_gemm_kernel
include/flux/cuda/gemm_impls/gemm_grouped_impl.hpp:108
Method
default_tile_scheduler
include/flux/cuda/gemm_impls/gemm_v3_impl.hpp:56
Method
derived
include/flux/cuda/gemm_impls/gemm_operator_base_default_impl.hpp:46
Method
do_epilogue
Perform epilogue computations and output
src/all_gather/sm80_all_gather_gemm.hpp:739
Method
do_epilogue
Perform epilogue computations and output
src/reduce_scatter/gemmk_universal_with_visitor_streamk.h:712
Method
elapsed_millis
Return the elapsed time (in milliseconds)
include/flux/cuda/cuda_common.h:195
Method
end_epilogue
/
src/reduce_scatter/epilogue_evt.hpp:328
Method
end_epilogue
src/reduce_scatter/epilogue_evt_nvshmem.hpp:306
Method
end_step
src/reduce_scatter/epilogue_evt.hpp:257
Method
end_step
src/reduce_scatter/epilogue_evt_nvshmem.hpp:264
Method
enumerate_split_meta_hparams_pairs
src/comm_none/gemm_v2_comm_none.hpp:210
Method
enumerate_split_meta_hparams_pairs
include/flux/gemm_operator_base.h:71
Method
eq
include/flux/flux.h:905
Function
estimate_gemm_sol_time_ms
refer to this: https://arnon.dk/matching-sm-architectures-arch-and-gencode-for-various-nvidia-cards/
python/flux/util.py:44
Function
exec_in_rank_order
(group: torch.distributed.ProcessGroup, func: Callable)
python/flux/dist_utils.py:79
Method
fetch_next_work
src/all_gather/sm90_all_gather_gemm_tma_warpspecialized_cooperative.hpp:773
Method
fetch_next_work
src/reduce_scatter/sm90_gemm_tma_warpspecialized_cooperative_reduce_scatter.hpp:735
Method
fields
include/flux/flux.h:955
Method
filter_fast_accum
include/flux/gemm_meta.h:386
Function
filter_kernel_schedule
include/flux/gemm_hparams.h:433
Method
filter_layout
include/flux/gemm_meta.h:401
Method
flags
src/reduce_scatter/reduce_scatter_kernel.hpp:310
Function
flux_create_tensor
src/ths_op/flux_shm.cc:170
Method
forward
( self, input: torch.Tensor, weight: torch.Tensor, bias: torch.Tensor )
python/flux/gemm_rs_sm80.py:135
Method
forward
src/all_gather/ths_op/all_gather_gemm_kernel_crossnode.cc:376
Method
forward
src/reduce_scatter/ths_op/gemm_reduce_scatter.cc:628
Method
forward_allgather
src/all_gather/ths_op/all_gather_gemm_kernel.cc:529
Method
forward_barrier
src/reduce_scatter/ths_op/gemm_reduce_scatter.cc:607
Method
forward_gemm
src/all_gather/ths_op/all_gather_gemm_kernel.cc:592
Method
forward_gemm
src/reduce_scatter/ths_op/gemm_reduce_scatter.cc:587
Method
forward_gemm_impl
src/all_gather/ths_op/all_gather_gemm_kernel.cc:440
Method
forward_gemm_impl
src/reduce_scatter/ths_op/gemm_reduce_scatter.cc:345
Method
forward_impl
src/all_gather/ths_op/all_gather_gemm_kernel_crossnode.cc:299
Method
forward_reduce_scatter
src/reduce_scatter/ths_op/gemm_reduce_scatter.cc:621
Method
forward_reduce_scatter_impl
src/reduce_scatter/ths_op/gemm_reduce_scatter.cc:481
Method
from_tuple
include/flux/flux.h:966
Method
from_tuple
include/flux/cuda/cutlass_v3_builder.hpp:96
Method
gemm
Executes one GEMM
src/all_gather/sm80_all_gather_gemm.hpp:848
Method
gemm
Executes one GEMM
src/reduce_scatter/gemmk_universal_with_visitor_streamk.h:829
Method
gemm_device
///////////////////// CRTP functions /////////////////////
include/flux/cuda/gemm_impls/gemm_grouped_impl.hpp:138
Method
gemm_device
///////////////////// CRTP functions /////////////////////
include/flux/cuda/gemm_impls/gemm_v2_impl.hpp:348
Method
gemm_kernel
src/comm_none/gemm_v2_comm_none.hpp:48
Method
gemm_kernel
src/all_gather/gemm_v2_ag_kernel.hpp:64
← previous
next →
401–500 of 784, ranked by callers