(
self,
in_channels: int,
prev_output_channel: int,
out_channels: int,
temb_channels: int,
resolution_idx: Optional[int] = None,
dropout: float = 0.0,
num_layers: int = 1,
resnet_eps: float = 1e-6,
resnet_time_scale_shift: str = "default",
resnet_act_fn: str = "swish",
resnet_groups: int = 32,
resnet_pre_norm: bool = True,
output_scale_factor: float = 1.0,
add_upsample: bool = True,
temporal_norm_num_groups: int = 32,
temporal_cross_attention_dim: Optional[int] = None,
temporal_num_attention_heads: int = 8,
temporal_max_seq_length: int = 32,
)
| 1442 | |
| 1443 | class UpBlockMotion(nn.Module): |
| 1444 | def __init__( |
| 1445 | self, |
| 1446 | in_channels: int, |
| 1447 | prev_output_channel: int, |
| 1448 | out_channels: int, |
| 1449 | temb_channels: int, |
| 1450 | resolution_idx: Optional[int] = None, |
| 1451 | dropout: float = 0.0, |
| 1452 | num_layers: int = 1, |
| 1453 | resnet_eps: float = 1e-6, |
| 1454 | resnet_time_scale_shift: str = "default", |
| 1455 | resnet_act_fn: str = "swish", |
| 1456 | resnet_groups: int = 32, |
| 1457 | resnet_pre_norm: bool = True, |
| 1458 | output_scale_factor: float = 1.0, |
| 1459 | add_upsample: bool = True, |
| 1460 | temporal_norm_num_groups: int = 32, |
| 1461 | temporal_cross_attention_dim: Optional[int] = None, |
| 1462 | temporal_num_attention_heads: int = 8, |
| 1463 | temporal_max_seq_length: int = 32, |
| 1464 | ): |
| 1465 | super().__init__() |
| 1466 | resnets = [] |
| 1467 | motion_modules = [] |
| 1468 | |
| 1469 | for i in range(num_layers): |
| 1470 | res_skip_channels = in_channels if (i == num_layers - 1) else out_channels |
| 1471 | resnet_in_channels = prev_output_channel if i == 0 else out_channels |
| 1472 | |
| 1473 | resnets.append( |
| 1474 | ResnetBlock2D( |
| 1475 | in_channels=resnet_in_channels + res_skip_channels, |
| 1476 | out_channels=out_channels, |
| 1477 | temb_channels=temb_channels, |
| 1478 | eps=resnet_eps, |
| 1479 | groups=resnet_groups, |
| 1480 | dropout=dropout, |
| 1481 | time_embedding_norm=resnet_time_scale_shift, |
| 1482 | non_linearity=resnet_act_fn, |
| 1483 | output_scale_factor=output_scale_factor, |
| 1484 | pre_norm=resnet_pre_norm, |
| 1485 | ) |
| 1486 | ) |
| 1487 | |
| 1488 | motion_modules.append( |
| 1489 | TransformerTemporalModel( |
| 1490 | num_attention_heads=temporal_num_attention_heads, |
| 1491 | in_channels=out_channels, |
| 1492 | norm_num_groups=temporal_norm_num_groups, |
| 1493 | cross_attention_dim=temporal_cross_attention_dim, |
| 1494 | attention_bias=False, |
| 1495 | activation_fn="geglu", |
| 1496 | positional_embeddings="sinusoidal", |
| 1497 | num_positional_embeddings=temporal_max_seq_length, |
| 1498 | attention_head_dim=out_channels // temporal_num_attention_heads, |
| 1499 | ) |
| 1500 | ) |
| 1501 |
nothing calls this directly
no test coverage detected