MCPcopy Create free account

hub / github.com/deepspeedai/DeepSpeed / functions

Functions10,922 in github.com/deepspeedai/DeepSpeed

↓ 4 callersMethodupdate
Updated the running absmax used to calculate params. Function Arguments : val : The __half2 value to update the running min and max with.
csrc/includes/quantization_utils.h:177
↓ 4 callersMethodupdate_difficulty
(self, global_steps)
deepspeed/runtime/data_pipeline/curriculum_scheduler.py:155
↓ 4 callersMethodupdate_state
csrc/includes/cpu_lion.h:55
↓ 4 callersFunctionvalidate_inferred_shape
Validate that the leading dim of the shard is of the expected size and aligns with the sharding logic for the attention computation itself.
tests/unit/inference/v2/model_implementations/sharding/test_qkv_sharding.py:31
↓ 4 callersMethodwait
(self, **kwargs)
deepspeed/runtime/zero/partition_parameters.py:72
↓ 4 callersFunctionwarp_uniform
csrc/deepspeed4science/evoformer_attn/gemm_kernel_utils.h:217
↓ 4 callersMethodzero_offload_param
(self)
deepspeed/runtime/engine.py:1128
↓ 4 callersMethodzero_reduce_bucket_size
(self)
deepspeed/runtime/engine.py:1167
↓ 3 callersMethodFloatTensor
(self, values)
tests/unit/v1/moe/test_autoep_unit.py:529
↓ 3 callersMethodGetMaxTokenLength
csrc/transformer/inference/includes/inference_context.h:183
↓ 3 callersMethodIntTensor
(self)
accelerator/hpu_accelerator.py:227
↓ 3 callersMethodQuantize
(self, value_list, quantize_bits, groups, key, merge_dim=0)
deepspeed/runtime/weight_quantizer.py:42
↓ 3 callersMethod__init__
(self, hidden_size=64)
tests/unit/v1/moe/test_autoep_unit.py:144
↓ 3 callersMethod__init__
(self, *args)
tests/unit/runtime/zero/test_zero_context_ancestry.py:46
↓ 3 callersMethod__init__
(self, num_layers=3)
deepspeed/sequence/test_autosp.py:74
↓ 3 callersMethod__init__
(self, params, lr=0.02, weight_decay=0, momentum=0.95, ns_method="gram")
deepspeed/runtime/zero/muon/original_muon.py:191
↓ 3 callersFunction_add
(graph, lhs, rhs, name, device_time=0)
tests/unit/compile/test_list_schedule.py:114
↓ 3 callersFunction_apply_dtype_to_config
Set bf16/fp16 in config_dict based on dtype; skip if not supported.
tests/unit/v1/zero/test_stage2_flatten_on_gpu.py:21
↓ 3 callersFunction_assert_forward_runs
(engine)
tests/unit/v1/moe/test_autoep_checkpoint.py:289
↓ 3 callersFunction_assert_no_secondary_tensor_group
(model: Module)
tests/unit/runtime/zero/test_zeropp.py:43
↓ 3 callersFunction_assert_nonzero_named_grad
(engine, *name_fragments)
tests/unit/v1/moe/test_autoep_autotp_runtime.py:147
↓ 3 callersFunction_assert_partition_status
(model: Module, valid_statuses: Set[ZeroParamStatus])
tests/unit/v1/zero/test_zero.py:503
↓ 3 callersFunction_assert_secondary_tensor_size
(model: Module)
tests/unit/runtime/zero/test_zeropp.py:56
↓ 3 callersMethod_autoep_expert_parallel_group
(self, params)
deepspeed/runtime/zero/stage3.py:626
↓ 3 callersFunction_autoep_expert_param_info
(autoep_metadata)
deepspeed/checkpoint/ds_to_universal.py:461
↓ 3 callersMethod_backward_epilogue
(self)
deepspeed/runtime/engine.py:2834
↓ 3 callersFunction_bias_activation_test_helper
Fully parameterized testing entry point.
tests/unit/inference/v2/kernels/core_ops/test_bias_activation.py:38
↓ 3 callersFunction_build_config
Partition config that matches q_proj and o_proj via regex.
tests/unit/module_inject/test_tp_partition_config_path.py:44
↓ 3 callersFunction_build_overlap_optimizer
(monkeypatch, *, resolves_data_dependency)
tests/unit/v1/zero/test_overlap_comm_record_stream.py:49
↓ 3 callersFunction_build_param_uc_restore_meta
Build the restore-facing parameter UC metadata. Restore metadata stays on the parameter object and may include details that are intentionally
deepspeed/module_inject/layers.py:60
↓ 3 callersFunction_capacity
(gates: Tensor, capacity_factor: Tensor, min_capacity: Tensor)
deepspeed/moe/sharded_moe.py:162
↓ 3 callersFunction_capture_matched_names
Run _replace_module and capture full_name values that match a spec.
tests/unit/module_inject/test_tp_partition_config_path.py:52
↓ 3 callersMethod_check_process_alive
Check if the checkpoint process is still alive. Note: Only call this when self.ckpt_process is not None. Some ranks don't have a chec
deepspeed/runtime/checkpoint_engine/decoupled_checkpoint_engine.py:120
↓ 3 callersFunction_collect_by_ep_rank
(local_tensor, ep_rank, ep_size, device)
tests/unit/v1/moe/test_autoep_checkpoint.py:155
↓ 3 callersFunction_compiler
(name)
tests/unit/compile/test_inductor_aot_kwargs.py:9
↓ 3 callersMethod_config
(self, zero_stage, buffer_dtype=None)
tests/unit/v1/half_precision/test_mixed_precision_dtype.py:103
↓ 3 callersFunction_config_dtype
(config)
tests/unit/v1/zero/test_zero_coalesce_grad_reduction.py:86
↓ 3 callersFunction_convert_checkpoint_to_universal
(save_dir, tag)
tests/unit/v1/moe/test_autoep_checkpoint.py:36
↓ 3 callersFunction_count_type
(cmds, classtype)
tests/unit/runtime/pipe/test_pipe_schedule.py:10
↓ 3 callersMethod_create_momentum_buffer
(self, num_elements, i, ds_id)
deepspeed/runtime/zero/stage3.py:983
↓ 3 callersMethod_device
(self)
tests/torch_compile/test_deepcompile_z3_release.py:25
↓ 3 callersFunction_dist_allgather_fn
(input_tensor: Tensor, output_tensor: Tensor, group=None)
deepspeed/runtime/zero/partition_parameters.py:108
↓ 3 callersFunction_drop_tokens
Divide a tensor among the tensor parallel ranks
deepspeed/moe/mappings.py:56
↓ 3 callersMethod_dump_state
(self)
deepspeed/io/fast_file_writer.py:174
↓ 3 callersMethod_enable_universal_checkpoint
(self)
deepspeed/runtime/bf16_optimizer.py:232
↓ 3 callersMethod_engine
(self, param_dtype=None, fp16=False, bf16=False)
tests/unit/v1/half_precision/test_mixed_precision_dtype.py:52
↓ 3 callersMethod_ensure_availability_of_partitioned_params
(self, params)
deepspeed/runtime/zero/partition_parameters.py:1613
↓ 3 callersFunction_expert_weight_parity_worker
(rank, world_size, tp_size, ep_size, zero_stage=0)
tests/unit/v1/moe/test_autoep_autotp_grad_parity.py:605
↓ 3 callersFunction_folded_zero2_config
(*, mixed_precision=True)
tests/unit/v1/moe/test_autoep_autotp_grad_parity.py:72
↓ 3 callersMethod_fp32_state_allgather
(self, param, fp32_state_partition)
deepspeed/runtime/zero/stage3.py:2764
↓ 3 callersMethod_fuse_lora_layer
(self, layer_id)
deepspeed/runtime/hybrid_engine.py:132
↓ 3 callersFunction_gather_optimizer_state_for_param
(engine, param, key)
tests/unit/v1/moe/test_autoep_checkpoint.py:199
↓ 3 callersFunction_gather_source_zero_params
Gather source ZeRO params while AutoEP reads full tensor values.
deepspeed/moe/ep_repack.py:25
↓ 3 callersFunction_gather_tokens
Gather tensors and concatenate them along a dimension
deepspeed/moe/mappings.py:30
↓ 3 callersFunction_gather_zero_param
(param)
tests/unit/v1/moe/test_autoep_checkpoint.py:150
↓ 3 callersFunction_get_autoep_metadata
(model_state)
deepspeed/checkpoint/ds_to_universal.py:441
↓ 3 callersMethod_get_buffer
(self, index)
deepspeed/runtime/swap_tensor/async_swapper.py:158
↓ 3 callersMethod_get_ckpt_name
(self, checkpoints_path, tag, mp_placeholder=None, pp_placeholder=None)
deepspeed/runtime/engine.py:4021
↓ 3 callersMethod_get_current_buffer
(self)
deepspeed/runtime/swap_tensor/utils.py:174
↓ 3 callersFunction_get_data_parallel_group
Get the data parallel group the caller rank belongs to.
deepspeed/utils/groups.py:701
↓ 3 callersFunction_get_expert_data_parallel_group
Get the expert data parallel group the caller rank belongs to.
deepspeed/utils/groups.py:638
↓ 3 callersFunction_get_expert_parallel_group
Get the expert parallel group the caller rank belongs to.
deepspeed/utils/groups.py:619
↓ 3 callersFunction_get_file_path
(tmpdir, file_prefix, index=0)
tests/unit/ops/aio/test_aio.py:38
↓ 3 callersFunction_get_local_rank
()
tests/unit/ops/aio/test_gds.py:26
↓ 3 callersMethod_get_model_type
Extract model type from module config or class name.
deepspeed/module_inject/auto_tp.py:480
↓ 3 callersMethod_get_optimizer_param
(self, param_name)
deepspeed/runtime/engine.py:3437
↓ 3 callersMethod_get_optimizer_state
(self, sd, state_key)
deepspeed/checkpoint/zero_checkpoint.py:120
↓ 3 callersFunction_get_test_write_file_and_device_buffer
(tmpdir, ref_buffer, gds_handle, index=0)
tests/unit/ops/aio/test_gds.py:52
↓ 3 callersFunction_get_test_write_file_and_pinned_tensor
(tmpdir, ref_buffer, aio_handle=None, index=0)
tests/unit/ops/aio/test_aio.py:62
↓ 3 callersFunction_get_test_write_file_and_unpinned_tensor
(tmpdir, ref_buffer, index=0)
tests/unit/ops/aio/test_aio.py:56
↓ 3 callersMethod_get_universal_checkpoint_info
(self)
deepspeed/runtime/bf16_optimizer.py:243
↓ 3 callersMethod_get_used_buffers
(self)
deepspeed/runtime/swap_tensor/utils.py:177
↓ 3 callersFunction_get_zero_param_intra_parallel_group
Get the ZeRO parameter partitioning intra parallel group the caller rank belongs to.
deepspeed/utils/groups.py:889
↓ 3 callersMethod_has_inf_or_nan
(x, j=None)
deepspeed/runtime/zero/stage_1_and_2.py:2409
↓ 3 callersFunction_init_engine
(config_dict, hidden_dim, seed=42)
tests/unit/v1/zero/test_zero2_offload_multi_backward.py:43
↓ 3 callersFunction_init_group_wise_weight_quantization
[Experimental] Apply group-wise weight quantization to model. Replace layers module according to config_list Args: model (nn.Module): A n
deepspeed/inference/quantization/quantization.py:20
↓ 3 callersMethod_is_grid_valid
(self)
deepspeed/runtime/pipe/topology.py:390
↓ 3 callersMethod_is_partition_offloaded
Whether the parameter partition needs the per-parameter offload path. The selective update assumes the partition is resident on the compute d
deepspeed/ops/adam/zenflow_torch_adam.py:403
↓ 3 callersMethod_link_all_hp_params
(self)
deepspeed/runtime/zero/stage_1_and_2.py:717
↓ 3 callersFunction_load_universal_dense_state
(universal_dir, param_name, key)
tests/unit/v1/moe/test_autoep_checkpoint.py:60
↓ 3 callersFunction_load_universal_expert_state
(universal_dir, param_name, key)
tests/unit/v1/moe/test_autoep_checkpoint.py:66
↓ 3 callersFunction_make_autoep_zero2_config
(ep_size)
tests/unit/v1/moe/test_autoep_grad_parity.py:50
↓ 3 callersFunction_make_freqs
(seq_len, rot_dim, theta=10000.0, device="cpu")
tests/unit/sequence/test_apply_rotary_pos_emb.py:13
↓ 3 callersMethod_make_key
(self, i, j)
deepspeed/checkpoint/reshape_meg_2d.py:52
↓ 3 callersFunction_make_spec
(**kwargs)
tests/unit/v1/moe/test_autoep_unit.py:66
↓ 3 callersFunction_mark_fake_zero_param
(param, full_data, partition_data=None, ds_id=0, name="param")
tests/unit/v1/moe/test_autoep_unit.py:101
↓ 3 callersMethod_mlp_gemm
(input, residual, input_bias, weight_interm, weight_out, bias, gamma, beta, eps, pre_layer_norm,
op_builder/npu/inference.py:214
↓ 3 callersMethod_model_parallel_all_reduce
Perform all reduce within model parallel group, if any.
deepspeed/runtime/zero/stage3.py:2180
↓ 3 callersMethod_model_parallel_all_reduce
Perform all reduce within model parallel group, if any.
deepspeed/runtime/zero/stage_1_and_2.py:1970
↓ 3 callersFunction_offloaded_stage3_param
(selected_indices)
tests/unit/ops/adam/test_zf_torch_adam.py:227
↓ 3 callersFunction_one_hot_to_float
(x, num_classes)
deepspeed/moe/sharded_moe.py:180
↓ 3 callersMethod_optimizer_has_ckpt_event_epilogue
(self)
deepspeed/runtime/engine.py:1443
↓ 3 callersMethod_optimizer_has_ckpt_event_prologue
(self)
deepspeed/runtime/engine.py:1440
↓ 3 callersMethod_overflow_check_and_loss_scale_update
(self)
deepspeed/runtime/zero/stage3.py:2487
↓ 3 callersFunction_patch_kwargs
(kwargs, monkeypatch)
tests/unit/compile/test_inductor_aot_kwargs.py:51
↓ 3 callersMethod_placeholder
(self, param)
tests/unit/ops/adam/test_zf_torch_adam.py:204
↓ 3 callersFunction_pre_ln_test_helper
(n_tokens: int, n_channels: int, dtype: torch.dtype, res_add: bool = False)
tests/unit/inference/v2/modules/test_cuda_pre_ln_module.py:36
↓ 3 callersFunction_pre_rms_test_helper
(n_tokens: int, n_channels: int, dtype: torch.dtype, res_add: bool = False)
tests/unit/inference/v2/modules/test_pre_rms_module.py:38
↓ 3 callersMethod_pre_step
(self)
deepspeed/runtime/zero/stage3.py:2305
↓ 3 callersMethod_qkv_gemm
(inputs, weight, q_scale, bias, gamma, beta, eps, add_bias, q_int8, transpose)
op_builder/npu/inference.py:58
← previousnext →801–900 of 10,922, ranked by callers