Prediction/evaluation loop, shared by :obj:`Trainer.evaluate()` and :obj:`Trainer.predict()`. Works both with or without labels.
(
self,
dataloader: DataLoader,
description: str,
metric_key_prefix: str = "eval",
)
| 791 | return images, weighted_loss, losses |
| 792 | |
| 793 | def evaluation_loop( |
| 794 | self, |
| 795 | dataloader: DataLoader, |
| 796 | description: str, |
| 797 | metric_key_prefix: str = "eval", |
| 798 | ) -> Tuple[Dict[str, float], int]: |
| 799 | """ |
| 800 | Prediction/evaluation loop, shared by :obj:`Trainer.evaluate()` and :obj:`Trainer.predict()`. |
| 801 | |
| 802 | Works both with or without labels. |
| 803 | """ |
| 804 | |
| 805 | batch_size = dataloader.batch_size |
| 806 | |
| 807 | logger.info(f"***** Running {description} *****") |
| 808 | if isinstance(dataloader.dataset, collections.abc.Sized): |
| 809 | logger.info(f" Num examples = {len(dataloader.dataset)}") |
| 810 | else: |
| 811 | logger.info(" Num examples: Unknown") |
| 812 | logger.info(f" Batch size = {batch_size}") |
| 813 | |
| 814 | self.model.eval() |
| 815 | |
| 816 | # Do this before wrapping. |
| 817 | eval_dataset = dataloader.dataset |
| 818 | |
| 819 | # Initialize containers |
| 820 | # losses/preds/labels on GPU/TPU (accumulated for eval_accumulation_steps) |
| 821 | prediction_outputs_host = None |
| 822 | # losses/preds/labels on CPU (final containers) |
| 823 | all_prediction_outputs = None |
| 824 | # Will be useful when we have an iterable dataset so don't know its length. |
| 825 | |
| 826 | # Main evaluation loop |
| 827 | for step, inputs in tqdm(enumerate(dataloader)): |
| 828 | # Prediction step |
| 829 | prediction_outputs = self.prediction_step(inputs) |
| 830 | |
| 831 | # Update containers on host |
| 832 | if prediction_outputs is not None: |
| 833 | prediction_outputs = distributed_concat(prediction_outputs) |
| 834 | prediction_outputs_host = ( |
| 835 | prediction_outputs if prediction_outputs_host is None else |
| 836 | nested_concat(prediction_outputs_host, prediction_outputs, padding_index=-100) |
| 837 | ) |
| 838 | |
| 839 | # Gather all tensors and put them back on the CPU if we have done enough accumulation steps. |
| 840 | if self.args.eval_accumulation_steps is not None and (step + 1) % self.args.eval_accumulation_steps == 0: |
| 841 | if prediction_outputs_host is not None: |
| 842 | prediction_outputs = nested_cpu(prediction_outputs_host) |
| 843 | all_prediction_outputs = ( |
| 844 | prediction_outputs if all_prediction_outputs is None else |
| 845 | nested_concat(all_prediction_outputs, prediction_outputs, padding_index=-100) |
| 846 | ) |
| 847 | |
| 848 | # Set back to None to begin a new accumulation |
| 849 | prediction_outputs_host = None |
| 850 |
nothing calls this directly
no test coverage detected