| 69 | class T5Attention(nn.Module): |
| 70 | |
| 71 | def __init__(self, dim, dim_attn, num_heads, dropout=0.1): |
| 72 | assert dim_attn % num_heads == 0 |
| 73 | super(T5Attention, self).__init__() |
| 74 | self.dim = dim |
| 75 | self.dim_attn = dim_attn |
| 76 | self.num_heads = num_heads |
| 77 | self.head_dim = dim_attn // num_heads |
| 78 | |
| 79 | # layers |
| 80 | self.q = nn.Linear(dim, dim_attn, bias=False) |
| 81 | self.k = nn.Linear(dim, dim_attn, bias=False) |
| 82 | self.v = nn.Linear(dim, dim_attn, bias=False) |
| 83 | self.o = nn.Linear(dim_attn, dim, bias=False) |
| 84 | self.dropout = nn.Dropout(dropout) |
| 85 | |
| 86 | def forward(self, x, context=None, mask=None, pos_bias=None): |
| 87 | """ |