| 557 | ) |
| 558 | |
| 559 | def forward(self, x, feat_cache=None, feat_idx=[0]): |
| 560 | |
| 561 | if feat_cache is not None: |
| 562 | idx = feat_idx[0] |
| 563 | cache_x = x[:, :, -CACHE_T:, :, :].clone() |
| 564 | if cache_x.shape[2] < 2 and feat_cache[idx] is not None: |
| 565 | cache_x = torch.cat( |
| 566 | [ |
| 567 | feat_cache[idx][:, :, -1, :, :].unsqueeze(2).to( |
| 568 | cache_x.device), |
| 569 | cache_x, |
| 570 | ], |
| 571 | dim=2, |
| 572 | ) |
| 573 | x = self.conv1(x, feat_cache[idx]) |
| 574 | feat_cache[idx] = cache_x |
| 575 | feat_idx[0] += 1 |
| 576 | else: |
| 577 | x = self.conv1(x) |
| 578 | |
| 579 | ## downsamples |
| 580 | for layer in self.downsamples: |
| 581 | if feat_cache is not None: |
| 582 | x = layer(x, feat_cache, feat_idx) |
| 583 | else: |
| 584 | x = layer(x) |
| 585 | |
| 586 | ## middle |
| 587 | for layer in self.middle: |
| 588 | if isinstance(layer, ResidualBlock) and feat_cache is not None: |
| 589 | x = layer(x, feat_cache, feat_idx) |
| 590 | else: |
| 591 | x = layer(x) |
| 592 | |
| 593 | ## head |
| 594 | for layer in self.head: |
| 595 | if isinstance(layer, CausalConv3d) and feat_cache is not None: |
| 596 | idx = feat_idx[0] |
| 597 | cache_x = x[:, :, -CACHE_T:, :, :].clone() |
| 598 | if cache_x.shape[2] < 2 and feat_cache[idx] is not None: |
| 599 | cache_x = torch.cat( |
| 600 | [ |
| 601 | feat_cache[idx][:, :, -1, :, :].unsqueeze(2).to( |
| 602 | cache_x.device), |
| 603 | cache_x, |
| 604 | ], |
| 605 | dim=2, |
| 606 | ) |
| 607 | x = layer(x, feat_cache[idx]) |
| 608 | feat_cache[idx] = cache_x |
| 609 | feat_idx[0] += 1 |
| 610 | else: |
| 611 | x = layer(x) |
| 612 | |
| 613 | return x |
| 614 | |
| 615 | |
| 616 | class Decoder3d(nn.Module): |