(self, src:Tensor, prev:Optional[Tensor]=None, key_padding_mask:Optional[Tensor]=None, attn_mask:Optional[Tensor]=None)
| 235 | |
| 236 | |
| 237 | def forward(self, src:Tensor, prev:Optional[Tensor]=None, key_padding_mask:Optional[Tensor]=None, attn_mask:Optional[Tensor]=None) -> Tensor: |
| 238 | |
| 239 | # Multi-Head attention sublayer |
| 240 | if self.pre_norm: |
| 241 | src = self.norm_attn(src) |
| 242 | ## Multi-Head attention |
| 243 | if self.res_attention: |
| 244 | src2, attn, scores = self.self_attn(src, src, src, prev, key_padding_mask=key_padding_mask, attn_mask=attn_mask) |
| 245 | else: |
| 246 | src2, attn = self.self_attn(src, src, src, key_padding_mask=key_padding_mask, attn_mask=attn_mask) |
| 247 | if self.store_attn: |
| 248 | self.attn = attn |
| 249 | ## Add & Norm |
| 250 | src = src + self.dropout_attn(src2) # Add: residual connection with residual dropout |
| 251 | if not self.pre_norm: |
| 252 | src = self.norm_attn(src) |
| 253 | |
| 254 | # Feed-forward sublayer |
| 255 | if self.pre_norm: |
| 256 | src = self.norm_ffn(src) |
| 257 | ## Position-wise Feed-Forward |
| 258 | src2 = self.ff(src) |
| 259 | ## Add & Norm |
| 260 | src = src + self.dropout_ffn(src2) # Add: residual connection with residual dropout |
| 261 | if not self.pre_norm: |
| 262 | src = self.norm_ffn(src) |
| 263 | |
| 264 | if self.res_attention: |
| 265 | return src, scores |
| 266 | else: |
| 267 | return src |
| 268 | |
| 269 | |
| 270 |
nothing calls this directly
no outgoing calls
no test coverage detected