(
module,
n_layer,
initializer_range=0.02, # Now only used for embedding layer.
rescale_prenorm_residual=True,
n_residuals_per_layer=1, # Change to 2 if we have MLP
)
| 26 | |
| 27 | # https://github.com/huggingface/transformers/blob/c28d04e9e252a1a099944e325685f14d242ecdcd/src/transformers/models/gpt2/modeling_gpt2.py#L454 |
| 28 | def _init_weights( |
| 29 | module, |
| 30 | n_layer, |
| 31 | initializer_range=0.02, # Now only used for embedding layer. |
| 32 | rescale_prenorm_residual=True, |
| 33 | n_residuals_per_layer=1, # Change to 2 if we have MLP |
| 34 | ): |
| 35 | with get_cuda_rng_tracker().fork(): |
| 36 | if isinstance(module, nn.Linear): |
| 37 | if not getattr(module.weight, "_no_reinit", False): |
| 38 | nn.init.normal_(module.weight, std=initializer_range) |
| 39 | if module.bias is not None: |
| 40 | if not getattr(module.bias, "_no_reinit", False): |
| 41 | nn.init.zeros_(module.bias) |
| 42 | elif isinstance(module, nn.Embedding): |
| 43 | nn.init.normal_(module.weight, std=initializer_range) |
| 44 | |
| 45 | for name, p in module.named_parameters(): |
| 46 | if name in ["in_proj.weight", "x_proj.weight", "conv1d.weight", "out_proj.weight"]: |
| 47 | nn.init.kaiming_uniform_(p, a=math.sqrt(5)) |
| 48 | |
| 49 | if rescale_prenorm_residual: |
| 50 | # Reinitialize selected weights subject to the OpenAI GPT-2 Paper Scheme: |
| 51 | # > A modified initialization which accounts for the accumulation on the residual path with model depth. Scale |
| 52 | # > the weights of residual layers at initialization by a factor of 1/√N where N is the # of residual layers. |
| 53 | # > -- GPT-2 :: https://openai.com/blog/better-language-models/ |
| 54 | # |
| 55 | # Reference (Megatron-LM): https://github.com/NVIDIA/Megatron-LM/blob/main/megatron/model/gpt_model.py |
| 56 | for name, p in module.named_parameters(): |
| 57 | if name in ["out_proj.weight", "fc2.weight"]: |
| 58 | # Special Scaled Initialization |
| 59 | nn.init.normal_( |
| 60 | p, |
| 61 | mean=0.0, |
| 62 | std=initializer_range / math.sqrt(n_residuals_per_layer * n_layer), |
| 63 | ) |
| 64 | |
| 65 | |
| 66 | @dataclass |
nothing calls this directly
no outgoing calls
no test coverage detected