| 300 | return TimestepEmbedSequential(operations.conv_nd(self.dims, channels, channels, 1, padding=0, dtype=dtype, device=device)) |
| 301 | |
| 302 | def forward(self, x, hint, timesteps, context, y=None, **kwargs): |
| 303 | t_emb = timestep_embedding(timesteps, self.model_channels, repeat_only=False).to(x.dtype) |
| 304 | emb = self.time_embed(t_emb) |
| 305 | |
| 306 | cond = kwargs["cond"] |
| 307 | num_video_frames = cond["num_video_frames"] |
| 308 | image_only_indicator = cond.get("image_only_indicator", None) |
| 309 | time_context = cond.get("time_context", None) |
| 310 | del cond |
| 311 | |
| 312 | guided_hint = self.input_hint_block(hint, emb, context, time_context=time_context, num_video_frames=num_video_frames, image_only_indicator=image_only_indicator) |
| 313 | |
| 314 | out_output = [] |
| 315 | out_middle = [] |
| 316 | |
| 317 | hs = [] |
| 318 | if self.num_classes is not None: |
| 319 | assert y.shape[0] == x.shape[0] |
| 320 | emb = emb + self.label_emb(y) |
| 321 | |
| 322 | h = x |
| 323 | for module, zero_conv in zip(self.input_blocks, self.zero_convs): |
| 324 | if guided_hint is not None: |
| 325 | h = module(h, emb, context, time_context=time_context, num_video_frames=num_video_frames, image_only_indicator=image_only_indicator) |
| 326 | h += guided_hint |
| 327 | guided_hint = None |
| 328 | else: |
| 329 | h = module(h, emb, context, time_context=time_context, num_video_frames=num_video_frames, image_only_indicator=image_only_indicator) |
| 330 | out_output.append(zero_conv(h, emb, context, time_context=time_context, num_video_frames=num_video_frames, image_only_indicator=image_only_indicator)) |
| 331 | |
| 332 | h = self.middle_block(h, emb, context, time_context=time_context, num_video_frames=num_video_frames, image_only_indicator=image_only_indicator) |
| 333 | out_middle.append(self.middle_block_out(h, emb, context, time_context=time_context, num_video_frames=num_video_frames, image_only_indicator=image_only_indicator)) |
| 334 | |
| 335 | return {"middle": out_middle, "output": out_output} |
| 336 | |
| 337 | |
| 338 | TEMPORAL_TRANSFORMER_BLOCKS = { |