(
self,
dim: int,
num_heads: int,
mlp_ratio: float = 4.0,
qkv_bias: bool = True,
proj_bias: bool = True,
ffn_bias: bool = True,
drop: float = 0.0,
attn_drop: float = 0.0,
init_values=None,
drop_path: float = 0.0,
act_layer: Callable[..., nn.Module] = nn.GELU,
norm_layer: Callable[..., nn.Module] = nn.LayerNorm,
ffn_layer: Callable[..., nn.Module] = Mlp,
qk_norm: bool = False,
rope=None,
kv_cache_sliding_window: int = 64,
kv_cache_scale_frames: int = 8,
kv_cache_cross_frame_special: bool = True,
kv_cache_include_scale_frames: bool = True,
kv_cache_camera_only: bool = False,
)
| 463 | """ |
| 464 | |
| 465 | def __init__( |
| 466 | self, |
| 467 | dim: int, |
| 468 | num_heads: int, |
| 469 | mlp_ratio: float = 4.0, |
| 470 | qkv_bias: bool = True, |
| 471 | proj_bias: bool = True, |
| 472 | ffn_bias: bool = True, |
| 473 | drop: float = 0.0, |
| 474 | attn_drop: float = 0.0, |
| 475 | init_values=None, |
| 476 | drop_path: float = 0.0, |
| 477 | act_layer: Callable[..., nn.Module] = nn.GELU, |
| 478 | norm_layer: Callable[..., nn.Module] = nn.LayerNorm, |
| 479 | ffn_layer: Callable[..., nn.Module] = Mlp, |
| 480 | qk_norm: bool = False, |
| 481 | rope=None, |
| 482 | kv_cache_sliding_window: int = 64, |
| 483 | kv_cache_scale_frames: int = 8, |
| 484 | kv_cache_cross_frame_special: bool = True, |
| 485 | kv_cache_include_scale_frames: bool = True, |
| 486 | kv_cache_camera_only: bool = False, |
| 487 | ) -> None: |
| 488 | super().__init__() |
| 489 | self.norm1 = norm_layer(dim) |
| 490 | self.attn = SDPAAttention( |
| 491 | dim=dim, num_heads=num_heads, qk_norm=qk_norm, qkv_bias=qkv_bias, |
| 492 | proj_bias=proj_bias, attn_drop=attn_drop, proj_drop=drop, rope=rope, |
| 493 | kv_cache_sliding_window=kv_cache_sliding_window, |
| 494 | kv_cache_scale_frames=kv_cache_scale_frames, |
| 495 | kv_cache_cross_frame_special=kv_cache_cross_frame_special, |
| 496 | kv_cache_include_scale_frames=kv_cache_include_scale_frames, |
| 497 | kv_cache_camera_only=kv_cache_camera_only, |
| 498 | ) |
| 499 | self.ls1 = LayerScale(dim, init_values=init_values) if init_values else nn.Identity() |
| 500 | self.drop_path1 = DropPath(drop_path) if drop_path > 0.0 else nn.Identity() |
| 501 | self.norm2 = norm_layer(dim) |
| 502 | self.mlp = ffn_layer(in_features=dim, hidden_features=int(dim * mlp_ratio), |
| 503 | act_layer=act_layer, drop=drop, bias=ffn_bias) |
| 504 | self.ls2 = LayerScale(dim, init_values=init_values) if init_values else nn.Identity() |
| 505 | self.drop_path2 = DropPath(drop_path) if drop_path > 0.0 else nn.Identity() |
| 506 | self.sample_drop_ratio = drop_path |
| 507 | |
| 508 | def forward(self, x: Tensor, pos=None, enable_ulysses_cp=False, |
| 509 | num_patches=None, num_special=None, num_frames=None, enable_3d_rope=False, |
nothing calls this directly
no test coverage detected