MCPcopy Create free account
hub / github.com/MotrixLab/AiOS / MSDeformAttn

Class MSDeformAttn

models/aios/ops/modules/ms_deform_attn.py:30–135  ·  view source on GitHub ↗

Source from the content-addressed store, hash-verified

28
29
30class MSDeformAttn(nn.Module):
31 def __init__(self, d_model=256, n_levels=4, n_heads=8, n_points=4, use_4D_normalizer=False):
32 """Multi-Scale Deformable Attention Module.
33
34 :param d_model hidden dimension
35 :param n_levels number of feature levels
36 :param n_heads number of attention heads
37 :param n_points number of sampling points per attention head per feature level
38 """
39 super().__init__()
40 if d_model % n_heads != 0:
41 raise ValueError('d_model must be divisible by n_heads, but got {} and {}'.format(d_model, n_heads))
42 _d_per_head = d_model // n_heads
43 # you'd better set _d_per_head to a power of 2 which is more efficient in our CUDA implementation
44 if not _is_power_of_2(_d_per_head):
45 warnings.warn("You'd better set d_model in MSDeformAttn to make the dimension of each attention head a power of 2 "
46 'which is more efficient in our CUDA implementation.')
47
48 self.im2col_step = 64
49
50 self.d_model = d_model
51 self.n_levels = n_levels
52 self.n_heads = n_heads
53 self.n_points = n_points
54
55 self.sampling_offsets = nn.Linear(d_model, n_heads * n_levels * n_points * 2)
56 self.attention_weights = nn.Linear(d_model, n_heads * n_levels * n_points)
57 self.value_proj = nn.Linear(d_model, d_model)
58 self.output_proj = nn.Linear(d_model, d_model)
59
60 self.use_4D_normalizer = use_4D_normalizer # false
61
62 self._reset_parameters()
63
64 def _reset_parameters(self):
65 constant_(self.sampling_offsets.weight.data, 0.)
66 thetas = torch.arange(self.n_heads, dtype=torch.float32) * (2.0 * math.pi / self.n_heads)
67 grid_init = torch.stack([thetas.cos(), thetas.sin()], -1)
68 grid_init = (grid_init / grid_init.abs().max(-1, keepdim=True)[0]).view(self.n_heads, 1, 1, 2).repeat(1, self.n_levels, self.n_points, 1)
69 for i in range(self.n_points):
70 grid_init[:, :, i, :] *= i + 1
71 with torch.no_grad():
72 self.sampling_offsets.bias = nn.Parameter(grid_init.view(-1))
73 constant_(self.attention_weights.weight.data, 0.)
74 constant_(self.attention_weights.bias.data, 0.)
75 xavier_uniform_(self.value_proj.weight.data)
76 constant_(self.value_proj.bias.data, 0.)
77 xavier_uniform_(self.output_proj.weight.data)
78 constant_(self.output_proj.bias.data, 0.)
79
80 def forward(self, query, reference_points, input_flatten, input_spatial_shapes, input_level_start_index, input_padding_mask=None):
81 """
82 :param query (N, Length_{query}, C)
83 :param reference_points (N, Length_{query}, n_levels, 2), range in [0, 1], top-left (0,0), bottom-right (1, 1), including padding area
84 or (N, Length_{query}, n_levels, 4), add additional (w, h) to form reference boxes
85 :param input_flatten (N, \sum_{l=0}^{L-1} H_l \cdot W_l, C)
86 :param input_spatial_shapes (n_levels, 2), [(H_0, W_0), (H_1, W_1), ..., (H_{L-1}, W_{L-1})]
87 :param input_level_start_index (n_levels, ), [0, H_0*W_0, H_0*W_0+H_1*W_1, H_0*W_0+H_1*W_1+H_2*W_2, ..., H_0*W_0+H_1*W_1+...+H_{L-1}*W_{L-1}]

Callers 2

__init__Method · 0.50
__init__Method · 0.50

Calls

no outgoing calls

Tested by

no test coverage detected