Set global variables, initialize distributed, and set autoresume and random seeds. `allow_no_cuda` should not be set unless using megatron for cpu only data processing. In general this arg should not be set unless you know what you are doing. Returns a function to finalize distri
(extra_args_provider=None, args_defaults={},
ignore_unknown_args=False, allow_no_cuda=False)
| 32 | |
| 33 | |
| 34 | def initialize_megatron(extra_args_provider=None, args_defaults={}, |
| 35 | ignore_unknown_args=False, allow_no_cuda=False): |
| 36 | """Set global variables, initialize distributed, and |
| 37 | set autoresume and random seeds. |
| 38 | `allow_no_cuda` should not be set unless using megatron for cpu only |
| 39 | data processing. In general this arg should not be set unless you know |
| 40 | what you are doing. |
| 41 | Returns a function to finalize distributed env initialization |
| 42 | (optionally, only when args.lazy_mpu_init == True) |
| 43 | |
| 44 | """ |
| 45 | if not allow_no_cuda: |
| 46 | # Make sure cuda is available. |
| 47 | assert torch.cuda.is_available(), 'Megatron requires CUDA.' |
| 48 | |
| 49 | # Parse args, build tokenizer, and set adlr-autoresume, |
| 50 | # tensorboard-writer, and timers. |
| 51 | set_global_variables(extra_args_provider=extra_args_provider, |
| 52 | args_defaults=args_defaults, |
| 53 | ignore_unknown_args=ignore_unknown_args) |
| 54 | |
| 55 | # torch.distributed initialization |
| 56 | def finish_mpu_init(): |
| 57 | args = get_args() |
| 58 | # Pytorch distributed. |
| 59 | _initialize_distributed() |
| 60 | |
| 61 | # Random seeds for reproducibility. |
| 62 | if args.rank == 0: |
| 63 | print('> setting random seeds to {} ...'.format(args.seed)) |
| 64 | _set_random_seed(args.seed) |
| 65 | |
| 66 | args = get_args() |
| 67 | if args.lazy_mpu_init: |
| 68 | args.use_cpu_initialization=True |
| 69 | # delayed initialization of DDP-related stuff |
| 70 | # We only set basic DDP globals |
| 71 | set_model_parallel_world_size(args.model_parallel_size) |
| 72 | # and return function for external DDP manager to call when it has DDP initialized |
| 73 | set_model_parallel_rank(args.rank) |
| 74 | return finish_mpu_init |
| 75 | else: |
| 76 | # Megatron's MPU is the master. Complete initialization right away. |
| 77 | finish_mpu_init() |
| 78 | |
| 79 | # Initialize memory buffers. |
| 80 | _initialize_mem_buffs() |
| 81 | |
| 82 | # Autoresume. |
| 83 | _init_autoresume() |
| 84 | |
| 85 | # Write arguments to tensorboard. |
| 86 | _write_args_to_tensorboard() |
| 87 | # No continuation function |
| 88 | return None |
| 89 | |
| 90 | |
| 91 | def setup_deepspeed_random_and_activation_checkpointing(args): |
no test coverage detected