r""" Transformer block used in [CogVideoX](https://github.com/THUDM/CogVideo) model. Parameters: dim (`int`): The number of channels in the input and output. num_attention_heads (`int`): The number of heads to use for multi-head attention. att
| 40 | |
| 41 | @maybe_allow_in_graph |
| 42 | class CogVideoXBlock(nn.Module): |
| 43 | r""" |
| 44 | Transformer block used in [CogVideoX](https://github.com/THUDM/CogVideo) model. |
| 45 | |
| 46 | Parameters: |
| 47 | dim (`int`): |
| 48 | The number of channels in the input and output. |
| 49 | num_attention_heads (`int`): |
| 50 | The number of heads to use for multi-head attention. |
| 51 | attention_head_dim (`int`): |
| 52 | The number of channels in each head. |
| 53 | time_embed_dim (`int`): |
| 54 | The number of channels in timestep embedding. |
| 55 | dropout (`float`, defaults to `0.0`): |
| 56 | The dropout probability to use. |
| 57 | activation_fn (`str`, defaults to `"gelu-approximate"`): |
| 58 | Activation function to be used in feed-forward. |
| 59 | attention_bias (`bool`, defaults to `False`): |
| 60 | Whether or not to use bias in attention projection layers. |
| 61 | qk_norm (`bool`, defaults to `True`): |
| 62 | Whether or not to use normalization after query and key projections in Attention. |
| 63 | norm_elementwise_affine (`bool`, defaults to `True`): |
| 64 | Whether to use learnable elementwise affine parameters for normalization. |
| 65 | norm_eps (`float`, defaults to `1e-5`): |
| 66 | Epsilon value for normalization layers. |
| 67 | final_dropout (`bool` defaults to `False`): |
| 68 | Whether to apply a final dropout after the last feed-forward layer. |
| 69 | ff_inner_dim (`int`, *optional*, defaults to `None`): |
| 70 | Custom hidden dimension of Feed-forward layer. If not provided, `4 * dim` is used. |
| 71 | ff_bias (`bool`, defaults to `True`): |
| 72 | Whether or not to use bias in Feed-forward layer. |
| 73 | attention_out_bias (`bool`, defaults to `True`): |
| 74 | Whether or not to use bias in Attention output projection layer. |
| 75 | """ |
| 76 | |
| 77 | def __init__( |
| 78 | self, |
| 79 | dim: int, |
| 80 | num_attention_heads: int, |
| 81 | attention_head_dim: int, |
| 82 | time_embed_dim: int, |
| 83 | dropout: float = 0.0, |
| 84 | activation_fn: str = "gelu-approximate", |
| 85 | attention_bias: bool = False, |
| 86 | qk_norm: bool = True, |
| 87 | norm_elementwise_affine: bool = True, |
| 88 | norm_eps: float = 1e-5, |
| 89 | final_dropout: bool = True, |
| 90 | ff_inner_dim: Optional[int] = None, |
| 91 | ff_bias: bool = True, |
| 92 | attention_out_bias: bool = True, |
| 93 | ): |
| 94 | super().__init__() |
| 95 | |
| 96 | # 1. Self Attention |
| 97 | self.norm1 = CogVideoXLayerNormZero(time_embed_dim, dim, norm_elementwise_affine, norm_eps, bias=True) |
| 98 | |
| 99 | self.attn1 = Attention( |