MCPcopy Create free account

hub / github.com/deepspeedai/DeepSpeed / functions

Functions10,922 in github.com/deepspeedai/DeepSpeed

↓ 5 callersFunctiontrain_cifar
(model, config, num_steps=400, average_dp_losses=True, fp16=True, seed=123)
tests/unit/alexnet_model.py:126
↓ 5 callersFunctiontranspose
(data)
deepspeed/module_inject/utils.py:9
↓ 5 callersMethodunscale_and_clip_grads
(self, grad_groups_flat, total_norm, apply_scale=True)
deepspeed/runtime/fp16/fused_optimizer.py:367
↓ 5 callersFunctionvalidate_folding_metadata
(metadata, *, tp_size,
deepspeed/checkpoint/autoep_universal.py:68
↓ 5 callersMethodwriter
(cls, path, dtype)
deepspeed/runtime/data_pipeline/data_sampling/indexed_dataset.py:375
↓ 5 callersFunctionzero3_linear_wrap
(input, weight, bias=None)
deepspeed/runtime/zero/linear.py:135
↓ 5 callersMethodzero_optimization_partition_gradients
(self)
deepspeed/runtime/engine.py:1176
↓ 4 callersMethodEvent
(self, enable_timing=True)
tests/unit/v1/compile/test_graph_profile.py:42
↓ 4 callersMethodInstance
(cls)
deepspeed/ops/transformer/inference/op_binding/workspace.py:32
↓ 4 callersMethodRATIO
csrc/includes/dropout.h:22
↓ 4 callersFunctionRun
csrc/includes/gemm_test.h:123
↓ 4 callersMethod__init__
(self, num_experts=4, ffn_hidden=128, hidden_size=64, intermediate_size=None)
tests/unit/v1/moe/autoep_test_utils.py:56
↓ 4 callersMethod__init__
(self, out=None, bias=None)
tests/unit/runtime/zero/test_zero_context_return.py:28
↓ 4 callersMethod__release_param
(self, param: Parameter, free_data: bool = True)
deepspeed/runtime/zero/partitioned_param_coordinator.py:624
↓ 4 callersMethod_aligned_size
(self, param)
deepspeed/runtime/zero/partition_parameters.py:1596
↓ 4 callersFunction_assert_params_match
(ref, test, label, tol=5e-5)
tests/unit/v1/zero/test_zero2_offload_multi_backward.py:58
↓ 4 callersMethod_assert_valid_mixed_precision_config
param_dtype, if set, must match the enabled fp16/bf16 mode. The optimizer/master-weight/reduction paths derive the model dtype from t
deepspeed/runtime/engine.py:1386
↓ 4 callersMethod_clear_fp32_optimizer_param_groups
(self)
deepspeed/runtime/zero/stage3.py:3083
↓ 4 callersMethod_configure_expert_parallel
Initialize AutoEP: detect MoE layers, create EP groups, replace with EP-enabled layers.
deepspeed/runtime/engine.py:535
↓ 4 callersMethod_create_zero_config
(self, hidden_dim, leaf_module=None)
tests/unit/runtime/zero/test_zero_leaf_module.py:300
↓ 4 callersFunction_do_parallel_work
(do_work, work_chunks, num_workers)
deepspeed/checkpoint/ds_to_universal.py:365
↓ 4 callersFunction_do_reshape
(src_3d, tgt_3d)
tests/unit/checkpoint/test_reshape_checkpoint.py:9
↓ 4 callersFunction_ds_initialize_for_param_partitioning_testing
(model: Module, cfg: dict)
tests/unit/v1/zero/test_zero.py:497
↓ 4 callersMethod_dump_mapping
(self, data_map, map_tag=None)
deepspeed/checkpoint/deepspeed_checkpoint.py:237
↓ 4 callersMethod_engine
(self, param_dtype=None, buffer_dtype=None, fp16=False, bf16=False)
tests/unit/v1/half_precision/test_mixed_precision_dtype.py:26
↓ 4 callersMethod_engine
(self, module)
tests/unit/v1/half_precision/test_mixed_precision_dtype.py:72
↓ 4 callersFunction_expert_params
(engine)
tests/unit/v1/moe/test_autoep_checkpoint.py:128
↓ 4 callersMethod_flush_gradient_swapper
(self, gradient_swapper)
deepspeed/runtime/swap_tensor/optimizer_utils.py:226
↓ 4 callersFunction_folding_spec
(mp_mode="tp", tp_size=2)
tests/unit/v1/moe/test_autoep_autotp_grad_parity.py:46
↓ 4 callersFunction_full_grad_by_suffix
(engine, suffix)
tests/unit/v1/moe/test_autoep_autotp_grad_parity.py:298
↓ 4 callersMethod_gather_view_and_storage
(self, shard, graph_id, ds_id)
tests/torch_compile/test_deepcompile_z3_release.py:45
↓ 4 callersMethod_get_fp32_grad_state_partition
(self, param, release_swap_buffers)
deepspeed/runtime/zero/stage3.py:2776
↓ 4 callersMethod_get_param_partition_rank
(self, param)
deepspeed/runtime/zero/stage3.py:597
↓ 4 callersFunction_grad_parity_metrics
(actual, expected)
tests/unit/v1/moe/test_autoep_autotp_grad_parity.py:403
↓ 4 callersMethod_init_dc
(self)
tests/torch_compile/test_deepcompile_z3_release.py:28
↓ 4 callersFunction_initialize_folded_engine
(*, zero_stage=0, ep_size=2, mixed_precision=True)
tests/unit/v1/moe/test_autoep_autotp_runtime.py:139
↓ 4 callersFunction_make_engine
()
tests/unit/runtime/rollout/test_hybrid_engine_rollout.py:20
↓ 4 callersFunction_make_node_meta
(node: Node, ds_id: int, comm: bool)
deepspeed/compile/fx.py:112
↓ 4 callersFunction_make_param
(numel, ds_persist=False)
tests/unit/v1/compile/test_selective_gather.py:40
↓ 4 callersFunction_make_tokenizer
()
tests/unit/runtime/rollout/test_hybrid_engine_rollout.py:27
↓ 4 callersFunction_model_config
()
tests/unit/inference/v2/model_implementations/test_exaone4_5.py:25
↓ 4 callersMethod_reassign_or_swap_out_partitioned_parameters
(self, sub_group_id)
deepspeed/runtime/zero/stage3.py:2525
↓ 4 callersMethod_register_param
(self, dc, graph_id, ds_id, shape, persistent=False)
tests/torch_compile/test_deepcompile_z3_release.py:33
↓ 4 callersFunction_rel_names
(repo: TmpRepo, tests)
ci/test_tests_fetcher.py:107
↓ 4 callersMethod_release
(self, view, graph_id, ds_id, n_users, synchronize=True)
tests/torch_compile/test_deepcompile_z3_release.py:54
↓ 4 callersFunction_resolve_expected_grad_dtype
(param)
deepspeed/compile/init_z3.py:24
↓ 4 callersMethod_run_git
(self, args: list[str])
ci/tests_fetcher.py:190
↓ 4 callersFunction_run_router_grad_boundary
(engine, *, logical_dp_world_size, logical_dp_rank, seed)
tests/unit/v1/moe/test_autoep_autotp_grad_parity.py:285
↓ 4 callersMethod_sample_top_p
Sample from logits with temperature and nucleus (top-p) filtering.
deepspeed/runtime/rollout/hybrid_engine_rollout.py:224
↓ 4 callersFunction_set_cuda_rng_state
Sets the random number generator state of the current GPU. Arguments: new_state (torch.ByteTensor): The desired state This function i
deepspeed/runtime/activation_checkpointing/checkpointing.py:91
↓ 4 callersMethod_set_fp32_optimizer_param_groups
(self)
deepspeed/runtime/zero/stage3.py:3077
↓ 4 callersMethod_set_param_uc_meta
(self, param, *, partition_ty
deepspeed/module_inject/layers.py:341
↓ 4 callersMethod_should_materialize_tp_partition
(self)
deepspeed/module_inject/layers.py:372
↓ 4 callersMethod_start_timers
(self, timer_names)
deepspeed/runtime/engine.py:3360
↓ 4 callersMethod_toy_model_config
(self, shard_size)
tests/unit/checkpoint/test_mics_optimizer.py:25
↓ 4 callersFunction_tp_consistent_input
(engine, *, seed=1234)
tests/unit/v1/moe/test_autoep_autotp_runtime.py:132
↓ 4 callersMethod_tp_partition
(self, params_list)
deepspeed/module_inject/layers.py:717
↓ 4 callersMethod_validate_autoep_folding_checkpoint_metadata
(state, *,
deepspeed/runtime/engine.py:3772
↓ 4 callersFunction_verify_continuous_increase
(values)
tests/unit/runtime/test_lr_schedulers.py:27
↓ 4 callersMethodadd_data
(self, pp_index, tp_index, data)
deepspeed/checkpoint/reshape_meg_2d.py:22
↓ 4 callersFunctionadd_postprocess
(graph: Graph, node: Node, fn: Callable[..., Any],
deepspeed/compile/fx.py:84
↓ 4 callersFunctionalign_up
csrc/deepspeed4science/evoformer_attn/gemm_kernel_utils.h:127
↓ 4 callersFunctionall_gather_dp_groups
(groups_flat, partitioned_param_groups, dp_process_group, start_alignment_factor, all
deepspeed/runtime/utils.py:1016
↓ 4 callersMethodallocate
(self, num_elems, count, dtype)
deepspeed/runtime/swap_tensor/utils.py:199
↓ 4 callersMethodallocate_all
(self, num_elems, dtype)
deepspeed/runtime/swap_tensor/utils.py:215
↓ 4 callersFunctionargs_from_dict
(tmpdir, config_dict)
tests/unit/simple_model.py:309
↓ 4 callersFunctionassignment_ordinals_by_expert
Return stable ordinals within each contiguous expert segment.
deepspeed/moe/ep_tp_dispatch.py:43
↓ 4 callersMethodautotuning_profile_model_info
(self)
deepspeed/runtime/engine.py:1066
↓ 4 callersMethodaverage_tensor
(self, tensor: torch.Tensor, communication_data_type: torch.dtype)
deepspeed/runtime/zero/stage_1_and_2.py:1277
↓ 4 callersMethodbuild
Build the stored specification.
deepspeed/runtime/pipe/module.py:69
↓ 4 callersFunctioncheck_and_handle_empty_buffer
( buffer_m: torch.Tensor, original_shape: torch.Size, original_size: int, worker_error: torch.
deepspeed/runtime/comm/utils.py:11
↓ 4 callersFunctioncheck_injection
(model)
tests/unit/inference/test_inference.py:249
↓ 4 callersMethodclear
Clear experiment queues, does not reset self.experiment_count
deepspeed/autotuning/scheduler.py:246
↓ 4 callersMethodclear
csrc/deepspeed4science/evoformer_attn/kernel_backward.h:1009
↓ 4 callersMethodclose
(self)
deepspeed/io/fast_file_writer.py:94
↓ 4 callersFunctioncompare_loss
(model_cls, enable, zero_stage, model_dtype,
tests/unit/v1/zero/test_zero_autocast.py:67
↓ 4 callersMethodconfig_tp_params
Configures the weight tensor for training with tensor parallelism. This includes enabling gradients and associating necessary methods
deepspeed/module_inject/layers.py:322
↓ 4 callersMethodconfigure
(self, comms_config)
deepspeed/utils/comms_logging.py:78
↓ 4 callersFunctioncreate_deepspeed_engine
(model_class, zero_stage, seed=42, gradient_accumulation_steps=1, **model_kwargs)
tests/unit/v1/zero/test_zero_user_backward.py:117
↓ 4 callersFunctioncreate_file
(filename, num_bytes)
deepspeed/nvme/test_ds_aio_utils.py:78
↓ 4 callersFunctioncreate_file
(filename, num_bytes)
csrc/aio/py_test/test_ds_aio_utils.py:78
↓ 4 callersFunctioncreate_filename
(folder, read_op, size, tid)
deepspeed/nvme/test_ds_aio_utils.py:73
↓ 4 callersFunctioncreate_filename
(folder, read_op, size, tid)
csrc/aio/py_test/test_ds_aio_utils.py:73
↓ 4 callersMethodcreate_op_builder
(self, class_name)
accelerator/hpu_accelerator.py:286
↓ 4 callersFunctioncreate_page_locked_tensor
(num_elem, use_accelerator, aio_handle=None)
deepspeed/nvme/test_ds_aio_utils.py:86
↓ 4 callersFunctioncreate_page_locked_tensor
(num_elem, use_accelerator, aio_handle=None)
csrc/aio/py_test/test_ds_aio_utils.py:86
↓ 4 callersFunctioncuFileGetErrorString
csrc/gds/py_lib/deepspeed_gds_utils.h:76
↓ 4 callersMethodcurriculum_enabled_legacy
(self)
deepspeed/runtime/engine.py:947
↓ 4 callersMethodcurriculum_learning_enabled
(self)
deepspeed/runtime/engine.py:965
↓ 4 callersFunctiondense_to_sparse
Converts dense matrix with explicit zeros to sparse matrix
tests/unit/ops/sparse_attention/test_sparse_attention.py:22
↓ 4 callersMethoddequantize
(self, tensor: Tensor, quant_scale: Tensor, quant_min: Tensor)
deepspeed/inference/quantization/utils.py:105
↓ 4 callersFunctiondp_index_to_str
(dp_index)
deepspeed/checkpoint/ds_to_universal.py:192
↓ 4 callersFunctionduration_to_string
(duration, units=None, precision=DEFAULT_PRECISION)
deepspeed/profiling/flops_profiler/profiler.py:1174
↓ 4 callersMethoddynamic_loss_scale
(self)
deepspeed/runtime/engine.py:1355
↓ 4 callersMethodeval_batch
Evaluate the pipeline on a batch of data from ``data_iter``. The engine will evaluate ``self.train_batch_size()`` total samples collec
deepspeed/runtime/pipe/engine.py:427
↓ 4 callersFunctionextract_tensors
Separate objects in list/tuple into tensors and non-tensors and create a mapping to enable re-aggregation. The order of tensors and non-tenso
deepspeed/runtime/activation_checkpointing/checkpointing.py:305
↓ 4 callersFunctionf
(size)
deepspeed/compile/profilers/comm_profile.py:147
↓ 4 callersFunctionfind_node_by_name
(gm: GraphModule, name: str)
deepspeed/compile/fx.py:167
↓ 4 callersMethodflatten_dense_tensors_aligned
(self, tensor_list, alignment, use_cpu_data=False)
deepspeed/runtime/zero/stage_1_and_2.py:1120
↓ 4 callersMethodflops_profiler_profile_step
(self)
deepspeed/runtime/engine.py:1019
← previousnext →601–700 of 10,922, ranked by callers