| 28 | |
| 29 | |
| 30 | class MSDeformAttn(nn.Module): |
| 31 | def __init__(self, d_model=256, n_levels=4, n_heads=8, n_points=4, use_4D_normalizer=False): |
| 32 | """Multi-Scale Deformable Attention Module. |
| 33 | |
| 34 | :param d_model hidden dimension |
| 35 | :param n_levels number of feature levels |
| 36 | :param n_heads number of attention heads |
| 37 | :param n_points number of sampling points per attention head per feature level |
| 38 | """ |
| 39 | super().__init__() |
| 40 | if d_model % n_heads != 0: |
| 41 | raise ValueError('d_model must be divisible by n_heads, but got {} and {}'.format(d_model, n_heads)) |
| 42 | _d_per_head = d_model // n_heads |
| 43 | # you'd better set _d_per_head to a power of 2 which is more efficient in our CUDA implementation |
| 44 | if not _is_power_of_2(_d_per_head): |
| 45 | warnings.warn("You'd better set d_model in MSDeformAttn to make the dimension of each attention head a power of 2 " |
| 46 | 'which is more efficient in our CUDA implementation.') |
| 47 | |
| 48 | self.im2col_step = 64 |
| 49 | |
| 50 | self.d_model = d_model |
| 51 | self.n_levels = n_levels |
| 52 | self.n_heads = n_heads |
| 53 | self.n_points = n_points |
| 54 | |
| 55 | self.sampling_offsets = nn.Linear(d_model, n_heads * n_levels * n_points * 2) |
| 56 | self.attention_weights = nn.Linear(d_model, n_heads * n_levels * n_points) |
| 57 | self.value_proj = nn.Linear(d_model, d_model) |
| 58 | self.output_proj = nn.Linear(d_model, d_model) |
| 59 | |
| 60 | self.use_4D_normalizer = use_4D_normalizer # false |
| 61 | |
| 62 | self._reset_parameters() |
| 63 | |
| 64 | def _reset_parameters(self): |
| 65 | constant_(self.sampling_offsets.weight.data, 0.) |
| 66 | thetas = torch.arange(self.n_heads, dtype=torch.float32) * (2.0 * math.pi / self.n_heads) |
| 67 | grid_init = torch.stack([thetas.cos(), thetas.sin()], -1) |
| 68 | grid_init = (grid_init / grid_init.abs().max(-1, keepdim=True)[0]).view(self.n_heads, 1, 1, 2).repeat(1, self.n_levels, self.n_points, 1) |
| 69 | for i in range(self.n_points): |
| 70 | grid_init[:, :, i, :] *= i + 1 |
| 71 | with torch.no_grad(): |
| 72 | self.sampling_offsets.bias = nn.Parameter(grid_init.view(-1)) |
| 73 | constant_(self.attention_weights.weight.data, 0.) |
| 74 | constant_(self.attention_weights.bias.data, 0.) |
| 75 | xavier_uniform_(self.value_proj.weight.data) |
| 76 | constant_(self.value_proj.bias.data, 0.) |
| 77 | xavier_uniform_(self.output_proj.weight.data) |
| 78 | constant_(self.output_proj.bias.data, 0.) |
| 79 | |
| 80 | def forward(self, query, reference_points, input_flatten, input_spatial_shapes, input_level_start_index, input_padding_mask=None): |
| 81 | """ |
| 82 | :param query (N, Length_{query}, C) |
| 83 | :param reference_points (N, Length_{query}, n_levels, 2), range in [0, 1], top-left (0,0), bottom-right (1, 1), including padding area |
| 84 | or (N, Length_{query}, n_levels, 4), add additional (w, h) to form reference boxes |
| 85 | :param input_flatten (N, \sum_{l=0}^{L-1} H_l \cdot W_l, C) |
| 86 | :param input_spatial_shapes (n_levels, 2), [(H_0, W_0), (H_1, W_1), ..., (H_{L-1}, W_{L-1})] |
| 87 | :param input_level_start_index (n_levels, ), [0, H_0*W_0, H_0*W_0+H_1*W_1, H_0*W_0+H_1*W_1+H_2*W_2, ..., H_0*W_0+H_1*W_1+...+H_{L-1}*W_{L-1}] |