Code
Hub
Workspaces
Following
Trending
Connect
MCP
copy
Create free account
hub
/
github.com/bytedance/flux
/ functions
Functions
784 in github.com/bytedance/flux
⨍
Functions
784
◇
Types & classes
309
↓ 1 callers
Function
tuple_for_each_impl
include/flux/flux.h:234
↓ 1 callers
Function
tuple_has_elem
include/flux/flux.h:446
↓ 1 callers
Function
tuple_transform_impl
include/flux/flux.h:218
↓ 1 callers
Function
tuple_unpack_cat_impl
include/flux/flux.h:412
Method
AGThreadblockSwizzleStreamKRankOffset
Constructor
src/all_gather/sm80_all_gather_gemm_threadblock_swizzle.hpp:48
Method
AppendCloseIpcMemHandleCallbackRAII
src/ths_op/ths_op.cc:97
Method
Arguments
Default Constructor
src/all_gather/sm80_all_gather_gemm.hpp:155
Method
Arguments
src/reduce_scatter/gemmk_universal_with_visitor_streamk.h:139
Method
CUTE_STATIC_ASSERT
TILE_N divides MMA_N bytedance::flux::StaticPrintInt<typename TiledCopyS2R::TiledNumThr{}>{}; bytedance::flux::StaticPrintInt<size<0>(typename TiledMm
src/reduce_scatter/epilogue_vectorized_reduce_scatter.hpp:250
Method
CUTE_STATIC_ASSERT
FIXME : l_coord
src/reduce_scatter/epilogue_nvshmem_reduce_scatter.hpp:302
Method
CheckFail
include/flux/flux.h:92
Method
CreateReduceScatterStream
src/reduce_scatter/ths_op/gemm_reduce_scatter.cc:199
Method
DistEnv
src/cuda/utils.cc:104
Method
DistEnvTP
src/ths_op/ths_op.cc:278
Method
DistEnvTPWithEP
src/ths_op/ths_op.cc:292
Method
DynamicLibrary
src/cuda/nvml_stub.cc:76
Method
DynamicLibrary
src/cuda/nvml_stub.cc:83
Method
DynamicLibrary
src/cuda/cuda_stub.cc:68
Method
DynamicLibrary
src/cuda/cuda_stub.cc:75
Method
FLUX_NAMED_TUPLE_DEFINE_FIELD
include/flux/gemm_meta.h:33
Method
FLUX_NAMED_TUPLE_DEFINE_FIELD
include/flux/gemm_meta.h:203
Method
FLUX_NAMED_TUPLE_DEFINE_FIELD
include/flux/gemm_meta.h:244
Method
FLUX_NAMED_TUPLE_DEFINE_FIELD
include/flux/gemm_hparams.h:49
Method
FLUX_NAMED_TUPLE_DEFINE_FIELD
include/flux/gemm_hparams.h:121
Method
FluxNamedTupleBase
include/flux/flux.h:998
Method
GpuTimer
Constructor
include/flux/cuda/cuda_common.h:170
Method
MoeArguments
src/ths_op/ths_op.cc:319
Method
OpRegistry
include/flux/op_registry.h:289
Method
OutputRankSwizzler
src/reduce_scatter/tile_scheduler/tile_mappings.hpp:53
Function
PYBIND11_MODULE
pynvshmem/src/module.cpp:6
Function
PYBIND11_MODULE
src/ths_op/ths_op.cc:420
Method
Params
Default constructor
src/all_gather/sm80_all_gather_gemm.hpp:361
Method
Params
Default constructor
src/reduce_scatter/gemmk_universal_with_visitor_streamk.h:308
Method
PersistentTileSchedulerSm90ReduceScatter
src/reduce_scatter/tile_scheduler/sm90_tile_scheduler_reduce_scatter.hpp:86
Method
PersistentTileSchedulerSm90ReduceScatterStreamK
src/reduce_scatter/tile_scheduler/sm90_tile_scheduler_reduce_scatter.hpp:154
Method
PyTuningRecord
src/ths_op/ths_op.cc:108
Method
ReduceScatterRing1dPullGemmk
src/reduce_scatter/reduce_scatter_kernel.hpp:1230
Method
ReduceScatterRing1dPushGemmk
src/reduce_scatter/reduce_scatter_kernel.hpp:1320
Method
ReduceScatterRing2dPull
src/reduce_scatter/reduce_scatter_kernel.hpp:511
Method
ReduceScatterRing2dPullGemmk
src/reduce_scatter/reduce_scatter_kernel.hpp:815
Method
ReduceScatterRing2dPullPerWarp
src/reduce_scatter/reduce_scatter_kernel.hpp:640
Method
ReduceScatterRing2dPullWithQueue
src/reduce_scatter/reduce_scatter_kernel.hpp:1081
Method
ReduceScatterRing2dPushGemmk
src/reduce_scatter/reduce_scatter_kernel.hpp:939
Method
Sm90AGKernelTileScheduler
src/all_gather/sm90_all_gather_gemm_tile_scheduler.hpp:128
Method
Sm90AuxStoreReduceScatter
src/reduce_scatter/sm90_epilogue_evt.hpp:164
Method
Sm90ReduceScatterDma
src/reduce_scatter/sm90_reduce_scatter_utils.hpp:209
Method
Symbol
src/cuda/nvml_stub.cc:88
Method
Symbol
src/cuda/cuda_stub.cc:84
Method
ThreadblockSwizzleStreamKRankOffset
Constructor
src/reduce_scatter/tile_scheduler/threadblock_swizzle.hpp:39
Method
ThreadblockSwizzleStreamKRankOffsetAcrossNode
Constructor
src/reduce_scatter/tile_scheduler/threadblock_swizzle_acrossnode.hpp:40
Method
ThsOpsInitRegistry
include/flux/ths_op/ths_op.h:120
Method
Timer
include/flux/utils.h:96
Method
TuningConfigGenerator
include/flux/op_registry.h:71
Method
TuningConfigRegistry
include/flux/op_registry.h:123
Method
VisitorAuxLoadGemmk
src/reduce_scatter/gemmk_visitor_load.hpp:97
Method
VisitorAuxStoreScatter
src/reduce_scatter/epilogue_evt.hpp:158
Method
VisitorAuxStoreScatterAccrossNode
src/reduce_scatter/epilogue_evt_nvshmem.hpp:131
Method
WorkIdxMSwizzler
src/reduce_scatter/tile_scheduler/tile_mappings.hpp:82
Method
__init__
( self, tp_group: dist.ProcessGroup, nnodes: int, max_m: int, n_dim: i
python/flux/gemm_rs_sm80.py:87
Method
__init__
(self, deterministic: bool = True)
python/flux/dist_utils.py:27
Method
__init__
( self, weight: torch.Tensor, tp_group, full_m: int, nnodes: int = 1,
python/flux/ag_gemm.py:69
Method
__init__
: ring_mode[int] -1 for auto: nvlink machine default to 0, non-nvlink machine default to 2 0: nvlink mode: all-to-all
python/flux/ag_kernel_crossnode.py:86
Method
__init__
(self, name: str, output: torch.Tensor, gemm_time_ms: float)
test/test_gemm_only.py:41
Method
__init__
( self, name: str, output: torch.Tensor, total_ms: float, time1: str,
test/test_ag_kernel_pyshmem.py:59
Method
__init__
( self, name: str, output: torch.Tensor, gemm_time_ms: float, comm_time_ms: float )
test/test_gemm_rs.py:62
Method
__init__
( self, name: str, output: torch.Tensor, total_ms: float, time1: str,
test/test_ag_kernel.py:60
Method
__init__
( self, name: str, output: torch.Tensor, total_ms: float, time1: str,
test/test_ag_kernel_crossnode.py:61
Method
__init__
(self, node: clang.cindex.Cursor)
pynvshmem/tools/autogen_pybind.py:120
Method
__repr__
(self)
test/test_gemm_only.py:46
Method
__repr__
(self)
test/test_ag_kernel_pyshmem.py:77
Method
__repr__
(self)
test/test_gemm_rs.py:70
Method
__repr__
(self)
test/test_ag_kernel.py:82
Method
__repr__
(self)
test/test_ag_kernel_crossnode.py:84
Method
_ensure_topo_initialized
src/all_gather/ths_op/all_gather_gemm_kernel_crossnode.cc:853
Method
_ensure_topo_initialized
src/reduce_scatter/ths_op/gemm_reduce_scatter.cc:751
Function
_get_text_from_extent
(extent)
pynvshmem/tools/autogen_pybind.py:254
Method
_to_argument
(dtype, name)
pynvshmem/tools/autogen_pybind.py:227
Method
_to_param
(dtype, name)
pynvshmem/tools/autogen_pybind.py:242
Method
acquire_accumulators
Acquire accumulators from peers
src/all_gather/sm80_all_gather_gemm.hpp:722
Method
acquire_accumulators
Acquire accumulators from peers
src/reduce_scatter/gemmk_universal_with_visitor_streamk.h:695
Function
add
src/reduce_scatter/reduce_scatter_kernel.hpp:178
Method
add
include/flux/op_registry.h:75
Method
add
include/flux/op_registry.h:95
Function
add_continous
src/reduce_scatter/reduce_scatter_kernel.hpp:270
Function
add_tensor
src/reduce_scatter/epilogue_nvshmem_reduce_scatter.hpp:71
Method
all_gemm_hparams
include/flux/gemm_operator_base.h:57
Method
all_gemm_metas
include/flux/gemm_operator_base.h:52
Method
arrive_inc
include/flux/cuda/system_barrier.hpp:105
Method
arrive_inc_get
include/flux/cuda/system_barrier.hpp:116
Method
arrive_inc_get
include/flux/cuda/system_barrier.hpp:154
Method
as_tuple
include/flux/flux.h:971
Method
async_load
include/flux/cuda/memory_utils.hpp:141
Method
async_load
include/flux/cuda/memory_utils.hpp:155
Method
async_load
include/flux/cuda/memory_utils.hpp:169
Function
atomic_add_dev
include/flux/cuda/cuda_common_device.hpp:149
Function
atomic_add_sys
include/flux/cuda/cuda_common_device.hpp:137
Function
atomic_load_acquire_dev
include/flux/cuda/cuda_common_device.hpp:132
Function
atomic_load_dev
include/flux/cuda/cuda_common_device.hpp:127
Function
atomic_store_release_dev
include/flux/cuda/cuda_common_device.hpp:121
Function
atomic_store_release_sys
include/flux/cuda/cuda_common_device.hpp:99
← previous
next →
301–400 of 784, ranked by callers