| 204 | |
| 205 | |
| 206 | def predict_next_token( |
| 207 | runtime, |
| 208 | model, |
| 209 | prompt, |
| 210 | io_obj=None, |
| 211 | target_tokens=None, |
| 212 | ): |
| 213 | inputs = runtime.tokenizer(prompt, return_tensors="pt") |
| 214 | if torch.cuda.is_available(): |
| 215 | inputs = inputs.to("cuda") |
| 216 | model = model.cuda() |
| 217 | else: |
| 218 | model = model |
| 219 | |
| 220 | model.eval() |
| 221 | outputs = model.generate( |
| 222 | inputs=inputs.input_ids, |
| 223 | attention_mask=inputs.attention_mask, |
| 224 | max_new_tokens=1, |
| 225 | num_beams=1, |
| 226 | pad_token_id=runtime.tokenizer.pad_token_id, |
| 227 | eos_token_id=runtime.tokenizer.eos_token_id, |
| 228 | return_dict_in_generate=True, |
| 229 | do_sample=False, |
| 230 | output_scores=True, |
| 231 | ) |
| 232 | |
| 233 | next_token_id = outputs.sequences[0, -1] |
| 234 | next_token = runtime.tokenizer.convert_ids_to_tokens(int(next_token_id)) |
| 235 | |
| 236 | output_text = runtime.tokenizer.decode( |
| 237 | outputs.sequences[0], |
| 238 | skip_special_tokens=True, |
| 239 | clean_up_tokenization_spaces=False, |
| 240 | ) |
| 241 | print(f"Next Token: {next_token}\n", file=io_obj) |
| 242 | print(f"Output Text:", file=io_obj) |
| 243 | print(output_text, file=io_obj) |
| 244 | |
| 245 | logits = outputs.scores[-1][0] |
| 246 | probs = torch.softmax(logits, dim=-1) |
| 247 | topk_tokens = torch.topk(probs, k=10) |
| 248 | topk_tokens_probs = topk_tokens.values |
| 249 | topk_tokens_ids = topk_tokens.indices |
| 250 | |
| 251 | print("\n\n", file=io_obj) |
| 252 | print("Top 10 Predictions:", file=io_obj) |
| 253 | for i, (tid, tp) in enumerate(zip(topk_tokens_ids, topk_tokens_probs)): |
| 254 | tok = runtime.tokenizer.convert_ids_to_tokens(int(tid)) |
| 255 | print(f"{i + 1}:\t{tok}\tp: {tp}") |
| 256 | |
| 257 | if target_tokens is not None: |
| 258 | target_token_indices = runtime.tokenizer.convert_tokens_to_ids(target_tokens) |
| 259 | target_probs = probs[target_token_indices] |
| 260 | topk_tokens = torch.topk(target_probs, k=len(target_token_indices)) |
| 261 | topk_tokens_probs = topk_tokens.values |
| 262 | topk_tokens_indices = topk_tokens.indices |
| 263 | |