(
self, hidden_states, attention_mask, rotary_pos_emb, kv_caches=None,
use_cache: Optional[bool] = True,
output_hidden_states: Optional[bool] = False,
)
| 516 | return self.layers[layer_number] |
| 517 | |
| 518 | def forward( |
| 519 | self, hidden_states, attention_mask, rotary_pos_emb, kv_caches=None, |
| 520 | use_cache: Optional[bool] = True, |
| 521 | output_hidden_states: Optional[bool] = False, |
| 522 | ): |
| 523 | if not kv_caches: |
| 524 | kv_caches = [None for _ in range(self.num_layers)] |
| 525 | presents = () if use_cache else None |
| 526 | if self.gradient_checkpointing and self.training: |
| 527 | if use_cache: |
| 528 | # logger.warning_once( |
| 529 | # "`use_cache=True` is incompatible with gradient checkpointing. Setting `use_cache=False`..." |
| 530 | # ) |
| 531 | use_cache = False |
| 532 | |
| 533 | all_self_attentions = None |
| 534 | all_hidden_states = () if output_hidden_states else None |
| 535 | for index in range(self.num_layers): |
| 536 | if output_hidden_states: |
| 537 | all_hidden_states = all_hidden_states + (hidden_states,) |
| 538 | |
| 539 | layer = self._get_layer(index) |
| 540 | if self.gradient_checkpointing and self.training: |
| 541 | layer_ret = torch.utils.checkpoint.checkpoint( |
| 542 | layer, |
| 543 | hidden_states, |
| 544 | attention_mask, |
| 545 | rotary_pos_emb, |
| 546 | kv_caches[index], |
| 547 | use_cache, |
| 548 | use_reentrant=False |
| 549 | ) |
| 550 | else: |
| 551 | layer_ret = layer( |
| 552 | hidden_states, |
| 553 | attention_mask, |
| 554 | rotary_pos_emb, |
| 555 | kv_cache=kv_caches[index], |
| 556 | use_cache=use_cache |
| 557 | ) |
| 558 | hidden_states, kv_cache = layer_ret |
| 559 | if use_cache: |
| 560 | presents = presents + (kv_cache,) |
| 561 | |
| 562 | if output_hidden_states: |
| 563 | all_hidden_states = all_hidden_states + (hidden_states,) |
| 564 | |
| 565 | # Final layer norm. |
| 566 | if self.post_layer_norm: |
| 567 | hidden_states = self.final_layernorm(hidden_states) |
| 568 | |
| 569 | return hidden_states, presents, all_hidden_states, all_self_attentions |
| 570 | |
| 571 | |
| 572 | class ChatGLMPreTrainedModel(PreTrainedModel): |
nothing calls this directly
no test coverage detected