r""" The approximate form of the Gaussian Error Linear Unit (GELU). For more details, see section 2 of this [paper](https://arxiv.org/abs/1606.08415). Parameters: dim_in (`int`): The number of channels in the input. dim_out (`int`): The number of channels in the output.
| 146 | |
| 147 | |
| 148 | class ApproximateGELU(nn.Module): |
| 149 | r""" |
| 150 | The approximate form of the Gaussian Error Linear Unit (GELU). For more details, see section 2 of this |
| 151 | [paper](https://arxiv.org/abs/1606.08415). |
| 152 | |
| 153 | Parameters: |
| 154 | dim_in (`int`): The number of channels in the input. |
| 155 | dim_out (`int`): The number of channels in the output. |
| 156 | bias (`bool`, defaults to True): Whether to use a bias in the linear layer. |
| 157 | """ |
| 158 | |
| 159 | def __init__(self, dim_in: int, dim_out: int, bias: bool = True): |
| 160 | super().__init__() |
| 161 | self.proj = nn.Linear(dim_in, dim_out, bias=bias) |
| 162 | |
| 163 | def forward(self, x: torch.Tensor) -> torch.Tensor: |
| 164 | x = self.proj(x) |
| 165 | return x * torch.sigmoid(1.702 * x) |