Print min, max, and norm of all parameters.
(optimizer, iteration)
| 106 | |
| 107 | |
| 108 | def print_params_min_max_norm(optimizer, iteration): |
| 109 | """Print min, max, and norm of all parameters.""" |
| 110 | index = 0 |
| 111 | rank = torch.distributed.get_rank() |
| 112 | string = "iteration, rank, index, tensor-model-parallel, min, max, norm\n" |
| 113 | optimizer_ = optimizer.optimizer |
| 114 | for param_group in optimizer_.param_groups: |
| 115 | for param in param_group["params"]: |
| 116 | index += 1 |
| 117 | min_ = param.data.min() |
| 118 | max_ = param.data.max() |
| 119 | norm = torch.linalg.norm(param.data) |
| 120 | string += "{:7d}, {:4d}, {:4d}, {:2d}, ".format( |
| 121 | iteration, rank, index, int(param.tensor_model_parallel) |
| 122 | ) |
| 123 | string += "{:.6E}, {:.6E}, {:.6E}\n".format(min_, max_, norm) |
| 124 | print(string, flush=True) |
| 125 | |
| 126 | |
| 127 | def check_adlr_autoresume_termination(iteration, model, optimizer, lr_scheduler): |
nothing calls this directly
no outgoing calls
no test coverage detected