Print loss after each step
(self, run_context)
| 59 | print("load has trained epoch :{} and step: {}".format(has_trained_epoch, has_trained_step), flush=True) |
| 60 | |
| 61 | def step_end(self, run_context): |
| 62 | """ |
| 63 | Print loss after each step |
| 64 | """ |
| 65 | cb_params = run_context.original_args() |
| 66 | if self._dataset_size > 0 and self.local_rank % 8 == 0: |
| 67 | percent, epoch_num = math.modf(cb_params.cur_step_num / |
| 68 | self._dataset_size) |
| 69 | if percent == 0: |
| 70 | epoch_num -= 1 |
| 71 | date = time.asctime(time.localtime(time.time())) |
| 72 | loss_value = cb_params.net_outputs[0].asnumpy() / self.micro_size |
| 73 | |
| 74 | if self.summary_writer is not None: |
| 75 | print(f"writing: {loss_value.item()}, {cb_params.net_outputs[2].asnumpy()}") |
| 76 | self.summary_writer.add_scalar( |
| 77 | tag="training_loss", |
| 78 | scalar_value=loss_value.item(), |
| 79 | global_step=cb_params.cur_step_num |
| 80 | + int(self.has_trained_step), |
| 81 | ) |
| 82 | self.summary_writer.add_scalar( |
| 83 | tag="loss_scale", |
| 84 | scalar_value=cb_params.net_outputs[2].asnumpy(), |
| 85 | global_step=cb_params.cur_step_num |
| 86 | + int(self.has_trained_step), |
| 87 | ) |
| 88 | print( |
| 89 | f"time: {date} local_rank: {int(self.local_rank)}, epoch: {int(epoch_num) + int(self.has_trained_epoch)}, step: {cb_params.cur_step_num + int(self.has_trained_step)}, output is {loss_value}, overflow is {cb_params.net_outputs[1].asnumpy()}, scale is {cb_params.net_outputs[2].asnumpy()}") |
| 90 | |
| 91 | |
| 92 | class EvalCallBack(Callback): |
nothing calls this directly
no outgoing calls
no test coverage detected