MCPcopy Create free account

hub / github.com/QDelta/Phantora / functions

Functions1,051 in github.com/QDelta/Phantora

↓ 2 callersFunctionnccl_group_start
()
phantora/cuda_call/src/capi.rs:850
↓ 2 callersMethodnccl_ungrouped_p2p
( &mut self, kind: P2PCallKind, host: HostId, curr_time: i64, count: u
phantora/phantora/src/simulator.rs:1107
↓ 2 callersMethodnetsim_run_one_step
(&mut self, until: Option<i64>)
phantora/phantora/src/event_queue.rs:255
↓ 2 callersMethodon_event
(&mut self, event: AppEvent)
phantora/netsim/src/app.rs:48
↓ 2 callersFunctionparse_device
(s: T)
phantora/phantora/src/torch_call.rs:100
↓ 2 callersMethodpercent
(&self)
phantora/visualizer/src/main.rs:87
↓ 2 callersMethodplink_probe_round
(&self, cur_ts: Timestamp)
phantora/netsim/src/background_flow.rs:114
↓ 2 callersMethodpush_flow
Add flow to a buffer of ready flows (`flow_bufs`) or emit the flow immediately
phantora/netsim/src/simulator.rs:949
↓ 2 callersMethodrecord_timeline_action
(&mut self, id: EventId, action: Action)
phantora/phantora/src/event_queue.rs:114
↓ 2 callersMethodregister_once
( &mut self, next_ready: Timestamp, token: Option<Token>, timer_id: Option<Tim
phantora/netsim/src/simulator.rs:247
↓ 2 callersMethodreverse
(&mut self)
phantora/netsim/src/simulator.rs:1185
↓ 2 callersMethodrun
(&mut self, niter: i32, call: &TorchCallInfo)
phantora/phantora/src/torch_estimate.rs:100
↓ 2 callersMethodrun_with_trace
(&mut self, trace: Trace)
phantora/netsim/src/simulator.rs:490
↓ 2 callersFunctionsend_to_simulator
(curr_sim_time: i64, call: CudaCall)
phantora/cuda_call/src/capi.rs:177
↓ 2 callersFunctionsimulator_socket_path
()
phantora/cuda_call/src/capi.rs:148
↓ 2 callersMethodsum
(iter: I)
phantora/netsim/src/bandwidth.rs:82
↓ 2 callersMethodtime_to_complete
(&self)
phantora/netsim/src/simulator.rs:1077
↓ 2 callersMethodtry_proceed_netsim
(&mut self, info: &str)
phantora/phantora/src/event_queue.rs:294
↓ 2 callersMethodupdate
(&mut self, rec: &Record)
phantora/visualizer/src/main.rs:163
↓ 1 callersFunction_analytic_permute
Reproduce torchtitan's _permute analytically for the load-balanced case, avoiding generate_permute_indices (a triton kernel that needs CUDA
tests/phantora_utils.py:279
↓ 1 callersFunction_init_device_props
stub/cudart.c:166
↓ 1 callersFunction_meta_list
(self, buffer)
tests/phantora_utils.py:161
↓ 1 callersMethodadd_end_time_to
(&mut self, id: EventId, new_end_time: i64)
phantora/phantora/src/event_queue.rs:476
↓ 1 callersMethodadd_flow
( &mut self, mut r: TraceRecord, cluster: &Cluster, sim_ts: Timestamp,
phantora/netsim/src/simulator.rs:912
↓ 1 callersFunctionalltoall_trace
(num_hosts: usize, flow_size: usize)
phantora/netsim/tests/alltoall.rs:138
↓ 1 callersMethodbcast
( &mut self, root: i32, ranks: &[(HostId, i32)], count: usize, dtype:
phantora/phantora/src/nccl_ops.rs:51
↓ 1 callersFunctionbuild_cloud
(setting: &Config)
phantora/netsim/src/config.rs:38
↓ 1 callersFunctionbuild_fatree_fake
(nports: usize, bw: Bandwidth, oversub_ratio: f64)
phantora/netsim/src/architecture.rs:175
↓ 1 callersFunctionbuild_fatree_fake
(nports: usize, bw: Bandwidth)
phantora/netsim/tests/alltoall.rs:99
↓ 1 callersFunctionbuild_pipeline_model
(args: argparse.Namespace)
tests/test_deepspeed.py:68
↓ 1 callersFunctionbuild_twolayer_multipath_cluster
( nspines: usize, nracks: usize, rack_size: usize, host_bw: Bandwidth, rack_uplink_port_bw
phantora/netsim/src/architecture.rs:229
↓ 1 callersMethodchange_start_time_notless
(&mut self, id: EventId, new_start_time: i64)
phantora/phantora/src/event_queue.rs:502
↓ 1 callersMethodcluster
(&mut self, cluster: Cluster)
phantora/netsim/src/simulator.rs:126
↓ 1 callersMethodcomplete_flows
(&mut self, sim_ts: Timestamp, ts_inc: Duration)
phantora/netsim/src/simulator.rs:973
↓ 1 callersMethodcompute_hash
(&mut self, flow: &Flow)
phantora/netsim/src/lib.rs:218
↓ 1 callersMethodcreate_ungrouped_p2p_endpoint
( &mut self, host: HostId, stream: CudaStream, curr_time: i64, count:
phantora/phantora/src/simulator.rs:1276
↓ 1 callersFunctioncudaGetDeviceCount
stub/cudart.c:62
↓ 1 callersFunctioncudaStreamCreate
stub/cudart.c:606
↓ 1 callersFunctioncudaStreamCreateWithPriority
stub/cudart.c:334
↓ 1 callersMethodcuda_add_latency
( &mut self, host: ResponseId, curr_time: i64, stream: CudaStream, lat
phantora/phantora/src/simulator.rs:782
↓ 1 callersFunctioncuda_device_reset
()
phantora/cuda_call/src/capi.rs:354
↓ 1 callersFunctioncuda_device_synchronize
(device: i32)
phantora/cuda_call/src/capi.rs:359
↓ 1 callersMethodcuda_device_synchronize
(&mut self, host: ResponseId, curr_time: i64, device: i32)
phantora/phantora/src/simulator.rs:629
↓ 1 callersFunctioncuda_event_query
( device: i32, stream: i32, id: i32, time_ref: *mut ffi::c_long, )
phantora/cuda_call/src/capi.rs:408
↓ 1 callersMethodcuda_event_query
(&mut self, host: ResponseId, curr_time: i64, event: CudaEvent)
phantora/phantora/src/simulator.rs:756
↓ 1 callersMethodcuda_event_record
(&mut self, host: ResponseId, curr_time: i64, event: CudaEvent)
phantora/phantora/src/simulator.rs:713
↓ 1 callersFunctioncuda_event_synchronize
(device: i32, stream: i32, id: i32)
phantora/cuda_call/src/capi.rs:398
↓ 1 callersMethodcuda_event_synchronize
(&mut self, host: ResponseId, curr_time: i64, event: CudaEvent)
phantora/phantora/src/simulator.rs:732
↓ 1 callersFunctioncuda_host_register
(ptr: usize, size: usize)
phantora/cuda_call/src/capi.rs:294
↓ 1 callersFunctioncuda_host_unregister
(ptr: usize)
phantora/cuda_call/src/capi.rs:299
↓ 1 callersFunctioncuda_launch_kernel
( func: *const ffi::c_void, // grid_dim: &Dim3, // block_dim: &Dim3, args: *mut *mut ffi::c_vo
phantora/cuda_call/src/capi.rs:474
↓ 1 callersFunctioncuda_mem_get_sizeinfo
(device: i32)
phantora/cuda_call/src/capi.rs:282
↓ 1 callersFunctioncuda_memcpy_async
( src: usize, dst: usize, size: usize, kind: i32, device: i32, stream: i32, )
phantora/cuda_call/src/capi.rs:304
↓ 1 callersMethodcuda_memcpy_async
( &mut self, host: ResponseId, curr_time: i64, size: usize, kind: Cuda
phantora/phantora/src/simulator.rs:603
↓ 1 callersFunctioncuda_register_free
(device: i32, ptr: usize)
phantora/cuda_call/src/capi.rs:273
↓ 1 callersFunctioncuda_register_malloc
(device: i32, ptr: usize, size: usize, total: usize)
phantora/cuda_call/src/capi.rs:257
↓ 1 callersFunctioncuda_stream_query
(device: i32, stream: i32)
phantora/cuda_call/src/capi.rs:446
↓ 1 callersMethodcuda_stream_query
(&mut self, host: ResponseId, curr_time: i64, stream: CudaStream)
phantora/phantora/src/simulator.rs:677
↓ 1 callersFunctioncuda_stream_synchronize
(device: i32, id: i32)
phantora/cuda_call/src/capi.rs:365
↓ 1 callersMethodcuda_stream_synchronize
(&mut self, host: ResponseId, curr_time: i64, stream: CudaStream)
phantora/phantora/src/simulator.rs:644
↓ 1 callersFunctioncuda_stream_wait_event
( stream_device: i32, stream_id: i32, event_device: i32, event_stream: i32, event_id: i32,
phantora/cuda_call/src/capi.rs:372
↓ 1 callersMethodcuda_stream_wait_event
( &mut self, host: ResponseId, curr_time: i64, stream: CudaStream, eve
phantora/phantora/src/simulator.rs:658
↓ 1 callersFunctioncurrent_sys_time_us
()
phantora/cuda_call/src/capi.rs:59
↓ 1 callersMethodemit_ready_flows
(&mut self, sim_ts: Timestamp)
phantora/netsim/src/simulator.rs:961
↓ 1 callersMethodenqueue_unmatched_p2p_endpoint
( &mut self, kind: P2PCallKind, key: P2PKey, endpoint: P2PFlowEndpoint, )
phantora/phantora/src/simulator.rs:1351
↓ 1 callersMethodestimate
(&mut self, call: &TorchCallInfo)
phantora/phantora/src/torch_estimate.rs:507
↓ 1 callersMethodestimate_sequence
(&mut self, calls: &[TorchCall])
phantora/phantora/src/torch_estimate.rs:517
↓ 1 callersMethodflash_attn
( &mut self, is_fwd: bool, is_bf16: bool, batch_size: i32, seqlen_q: i
phantora/phantora/src/cuda_estimate.rs:280
↓ 1 callersMethodflash_attn_call
( &mut self, host: ResponseId, curr_time: i64, call: CudaCall, stream:
phantora/phantora/src/simulator.rs:805
↓ 1 callersMethodget_cuda_stream
(&self)
phantora/cuda_call/src/lib.rs:222
↓ 1 callersFunctionget_optimizer
(model)
tests/test_megatron.py:199
↓ 1 callersMethodget_source
(&self, ix: LinkIx)
phantora/netsim/src/cluster.rs:346
↓ 1 callersMethodgpu_index
(&self)
phantora/phantora/src/torch_call.rs:131
↓ 1 callersMethodgroup_end
(&mut self)
phantora/cuda_call/src/capi.rs:749
↓ 1 callersMethodgroup_start
(&mut self)
phantora/cuda_call/src/capi.rs:745
↓ 1 callersMethodhandle_cuda_call
(&mut self, msg: CudaCallMsg)
phantora/phantora/src/simulator.rs:450
↓ 1 callersMethodhandle_exit
(&mut self, host: ResponseId, curr_time: i64)
phantora/phantora/src/simulator.rs:1594
↓ 1 callersMethodhandle_torch_call
(&mut self, call: TorchCall)
phantora/phantora/src/simulator.rs:1582
↓ 1 callersFunctionhash_vm_pair
(f: &Flow)
phantora/netsim/src/simulator.rs:587
↓ 1 callersMethodhost_mapping
(&mut self, host_mapping: Vec<String>)
phantora/netsim/src/simulator.rs:115
↓ 1 callersFunctioninstall_phantora_deepspeed_patches
Patch DeepSpeed PP tensor metadata exchange. Used by tests/test_deepspeed.py after deepspeed.init_distributed(). Patch targets: - deep
tests/phantora_utils.py:93
↓ 1 callersFunctioninstall_phantora_gpt_oss_patches
Make HF gpt-oss MoE runnable under Phantora's payload-free simulation. Used by tests/test_deepspeed.py when --model gpt_oss. gpt-oss's GptOs
tests/phantora_utils.py:602
↓ 1 callersFunctioninstall_phantora_megatron_moe_patches
Make Megatron-core MoE runnable under Phantora's payload-free simulation. Used by tests/test_megatron.py when num_moe_experts is set. Assumes exp
tests/phantora_utils.py:446
↓ 1 callersFunctioninstall_phantora_torchtitan_moe_patches
Make TorchTitan (>=0.2.0) MoE runnable under Phantora's payload-free sim. Must run BEFORE the model is built (the dispatch hook and GroupedExpert
tests/phantora_utils.py:212
↓ 1 callersFunctioninstall_phantora_torchtitan_patches
Patch TorchTitan PP shape-inference metadata exchange. Used by tests/test_torchtitan.py before constructing TorchTitan Trainer. Patch target
tests/phantora_utils.py:348
↓ 1 callersMethodinto_call
(self, arg_device: i32)
phantora/phantora/src/torch_call.rs:177
↓ 1 callersMethodinto_msg
(self)
phantora/phantora/src/torch_call.rs:16
↓ 1 callersMethodkbps
(self)
phantora/netsim/src/bandwidth.rs:50
↓ 1 callersFunctionkind_range
(kind: Kind)
phantora/phantora/src/torch_estimate.rs:18
↓ 1 callersFunctionmain
(args: argparse.Namespace)
tests/test_deepspeed.py:218
↓ 1 callersFunctionmain
( tensor_parallel_size, pipeline_model_parallel_size, virtual_pipeline_model_parallel_size, nu
tests/test_megatron.py:253
↓ 1 callersFunctionmain_loop
()
phantora/phantora/src/main.rs:35
↓ 1 callersFunctionmask_process
(pid: u32, cores: &[usize])
phantora/phantora/src/simulator.rs:18
↓ 1 callersMethodmax_min_fairness_converge
(&mut self)
phantora/netsim/src/simulator.rs:319
↓ 1 callersMethodmbps
(self)
phantora/netsim/src/bandwidth.rs:56
↓ 1 callersMethodmemcpy
(&mut self, kind: CudaMemcpyKind, size: usize)
phantora/phantora/src/cuda_estimate.rs:262
↓ 1 callersFunctionnccl_all_gather
( count: usize, dtype: i32, comm_id: *const ffi::c_char, rank: i32, device: i32, strea
phantora/cuda_call/src/capi.rs:945
↓ 1 callersFunctionnccl_all_reduce
( count: usize, dtype: i32, op: i32, comm_id: *const ffi::c_char, rank: i32, device: i
phantora/cuda_call/src/capi.rs:923
↓ 1 callersFunctionnccl_bcast
( count: usize, dtype: i32, root: i32, comm_id: *const ffi::c_char, rank: i32, device:
phantora/cuda_call/src/capi.rs:901
↓ 1 callersMethodnccl_bcast
( &mut self, host: ResponseId, curr_time: i64, count: usize, dtype: Nc
phantora/phantora/src/simulator.rs:1550
← previousnext →101–200 of 1,051, ranked by callers