Code
Hub
Workspaces
Following
Trending
Connect
MCP
copy
Create free account
hub
/
github.com/bytedance/flux
/ functions
Functions
784 in github.com/bytedance/flux
⨍
Functions
784
◇
Types & classes
309
Method
is_C_load_needed
src/reduce_scatter/sm90_epilogue_evt.hpp:178
Function
lazy_init_buffer_tensor
src/ths_op/ths_op.cc:405
Method
ld_acquire
Load flag, as a strong acquire operation (int specialization)
include/flux/cuda/system_barrier.hpp:36
Method
less
include/flux/flux.h:894
Function
load_tune_config_from_file
src/cuda/op_registry_proto_utils.cc:165
Function
load_tuning_record
src/ths_op/ths_op.cc:112
Function
local_rank_to_global_rank
include/flux/utils.h:36
Method
local_red
include/flux/cuda/memory_utils.hpp:193
Method
local_red
include/flux/cuda/memory_utils.hpp:223
Method
lower_name
include/flux/flux.h:945
Function
main
src/comm_none/test/test_gemm_only.cc:86
Function
main
src/reduce_scatter/test/test_gemm_rs.cc:142
Method
make_gather_rs_meta
include/flux/gemm_meta.h:259
Method
make_gemm_meta
include/flux/gemm_meta.h:354
Function
make_index_seq_tuple
include/flux/flux.h:328
Method
make_space_gemm_meta
include/flux/gemm_meta.h:424
Function
memset_continous_kernel
src/reduce_scatter/reduce_scatter_kernel.hpp:239
Method
num_fields
include/flux/flux.h:950
Function
nvshmemi_bootstrap_plugin_init
pynvshmem/bootstrap/torch/bootstrap_torch.cpp:122
Function
offset_index_sequence
include/flux/flux.h:157
Method
operator()
src/all_gather/sm80_all_gather_gemm.hpp:1028
Method
operator()
src/all_gather/sm90_all_gather_gemm_tma_warpspecialized_cooperative.hpp:387
Method
operator()
src/reduce_scatter/epilogue_reduce_scatter.hpp:159
Method
operator()
Runs the kernel using initialized state.
src/reduce_scatter/gemm_v2_reduce_scatter.hpp:116
Method
operator()
src/reduce_scatter/reduce_scatter_kernel.hpp:164
Method
operator()
src/reduce_scatter/reduce_scatter_kernel.hpp:2004
Method
operator()
src/reduce_scatter/epilogue_vectorized_reduce_scatter.hpp:170
Method
operator()
src/reduce_scatter/epilogue_nvshmem_reduce_scatter.hpp:182
Method
operator()
src/reduce_scatter/gemmk_universal_with_visitor_streamk.h:963
Method
operator()
src/reduce_scatter/sm90_gemm_tma_warpspecialized_cooperative_reduce_scatter.hpp:335
Function
operator<
include/flux/flux.h:918
Method
operator=
src/cuda/nvml_stub.cc:77
Method
operator=
src/cuda/cuda_stub.cc:69
Function
operator==
include/flux/flux.h:925
Method
overflow
src/cuda/utils.cc:27
Function
pad_to
pad `sz` to the minimum multiple of `pad`
include/flux/flux.h:623
Method
partition
src/reduce_scatter/visitor_2x_bsr.hpp:140
Function
pathlib_wrapper
(func)
setup.py:80
Function
print_cute_per_kernel
include/flux/cuda/cuda_common_device.hpp:81
Function
print_per_block_
include/flux/cuda/cuda_common_device.hpp:61
Method
process_tile
src/all_gather/sm80_all_gather_gemm.hpp:787
Method
process_tile
src/reduce_scatter/gemmk_universal_with_visitor_streamk.h:768
Method
profiling
src/comm_none/ths_op/gemm_only.cc:274
Function
ptr_offset
calculate void* ptr with offset of bytes
include/flux/flux.h:612
Function
put_tensor_on_stream
( dst_tensor: torch.Tensor, src_tensor: torch.Tensor, peer: int, stream: CUDA_STREAM_TYPE, )
python/flux/pynvshmem/_wrapper.py:20
Function
putmem_on_stream
pynvshmem/src/functions.cpp:80
Method
pybind_name
(self)
pynvshmem/tools/autogen_pybind.py:148
Method
pybind_name_wrapper
(self)
pynvshmem/tools/autogen_pybind.py:154
Function
pyflux_barrier_all_on_stream
src/ths_op/flux_shm.cc:216
Function
quiet
pynvshmem/src/functions.cpp:61
Function
rank_shift
include/flux/utils.h:82
Method
recast_backto_bf16
src/reduce_scatter/epilogue_evt_nvshmem.hpp:247
Method
red_release
Reduce into flag, with release pattern (int specialization)
include/flux/cuda/system_barrier.hpp:46
Method
reduce_buffer_tile
src/reduce_scatter/reduce_scatter_kernel.hpp:368
Method
register_creator
include/flux/op_registry.h:140
Method
register_dispatcher
include/flux/op_registry.h:173
Function
releaseCpuDataCallback
src/ths_op/ths_op.cc:89
Method
reset_signals
src/all_gather/ths_op/all_gather_gemm_kernel_crossnode.cc:381
Function
rfind
(string, pattern)
pynvshmem/tools/autogen_pybind.py:111
Method
ring_all_gather
src/all_gather/ths_op/all_gather_gemm_kernel.cc:889
Method
ring_all_gather
src/all_gather/ths_op/all_gather_gemm_kernel_crossnode.cc:808
Method
ring_all_gather_reorder
src/all_gather/ths_op/all_gather_gemm_kernel.cc:895
Method
ring_all_gather_reorder
src/all_gather/ths_op/all_gather_gemm_kernel_crossnode.cc:814
Method
ring_all_gather_run
src/all_gather/ths_op/all_gather_gemm_kernel.cc:907
Method
ring_all_gather_run
src/all_gather/ths_op/all_gather_gemm_kernel_crossnode.cc:826
Method
run
Runs the kernel using initialized state.
src/reduce_scatter/gemm_v2_reduce_scatter.hpp:82
Method
run
src/reduce_scatter/reduce_scatter_kernel.hpp:514
Method
run
src/reduce_scatter/reduce_scatter_kernel.hpp:643
Method
run
src/reduce_scatter/reduce_scatter_kernel.hpp:818
Method
run
src/reduce_scatter/reduce_scatter_kernel.hpp:942
Method
run
src/reduce_scatter/reduce_scatter_kernel.hpp:1084
Method
run
src/reduce_scatter/reduce_scatter_kernel.hpp:1233
Method
run
src/reduce_scatter/reduce_scatter_kernel.hpp:1323
Method
run
include/flux/cuda/gemm_impls/gemm_operator_base_default_impl.hpp:55
Function
run_gemm_only
src/comm_none/test/test_gemm_only.cc:32
Function
run_per_segment_kernel
src/reduce_scatter/reduce_scatter_kernel.hpp:1443
Function
run_per_segment_kernel_multinode
src/reduce_scatter/reduce_scatter_kernel.hpp:1646
Function
run_per_segment_kernel_tp8
src/reduce_scatter/reduce_scatter_kernel.hpp:1508
Function
run_per_segment_with_cudaMemcpyAsync
src/reduce_scatter/reduce_scatter_kernel.hpp:1863
Method
run_per_segment_with_cuda_core
src/reduce_scatter/reduce_scatter_kernel.hpp:2009
Function
run_perf
(expname, warmups, iters, func, sync_per_iter=False)
test/utils.py:106
Method
separate_reduction
src/all_gather/sm80_all_gather_gemm.hpp:749
Method
separate_reduction
src/reduce_scatter/gemmk_universal_with_visitor_streamk.h:730
Method
set_ready
src/all_gather/ths_op/transfers.hpp:28
Method
set_ready
src/all_gather/ths_op/all_gather_gemm_kernel.cc:798
Method
share_accumulators
Share accumulators with peers
src/all_gather/sm80_all_gather_gemm.hpp:688
Method
share_accumulators
Share accumulators with peers
src/reduce_scatter/gemmk_universal_with_visitor_streamk.h:661
Function
sleep_async
src/reduce_scatter/reduce_scatter_kernel.hpp:208
Method
split_slice_meta
include/flux/gemm_operator_base.h:63
Method
sync
src/reduce_scatter/epilogue_nvshmem_reduce_scatter.hpp:203
Method
sync
include/flux/cuda/system_barrier.hpp:27
Method
tid2coord
src/reduce_scatter/visitor_2x_bsr.hpp:133
Method
toString
src/ths_op/ths_op.cc:281
Method
to_argument_type
src/reduce_scatter/gemm_v2_reduce_scatter.hpp:484
Function
to_cuda_dtype
include/flux/cuda/cuda_common.h:75
Method
to_fp8_gemm_args_impl
src/comm_none/gemm_v2_comm_none.hpp:54
Method
to_fp8_gemm_args_impl
src/reduce_scatter/gemm_v2_reduce_scatter.hpp:389
Method
to_gather_rs_meta
include/flux/gemm_meta.h:265
Method
to_gemm_args
src/comm_none/gemm_v3_comm_none.hpp:126
Method
to_gemm_args
src/comm_none/gemm_v2_comm_none.hpp:163
← previous
next →
601–700 of 784, ranked by callers