MCPcopy Create free account
hub / github.com/THUDM/LongWriter / forward

Method forward

train/patch/modeling_chatglm.py:518–569  ·  view source on GitHub ↗
(
            self, hidden_states, attention_mask, rotary_pos_emb, kv_caches=None,
            use_cache: Optional[bool] = True,
            output_hidden_states: Optional[bool] = False,
    )

Source from the content-addressed store, hash-verified

516 return self.layers[layer_number]
517
518 def forward(
519 self, hidden_states, attention_mask, rotary_pos_emb, kv_caches=None,
520 use_cache: Optional[bool] = True,
521 output_hidden_states: Optional[bool] = False,
522 ):
523 if not kv_caches:
524 kv_caches = [None for _ in range(self.num_layers)]
525 presents = () if use_cache else None
526 if self.gradient_checkpointing and self.training:
527 if use_cache:
528 # logger.warning_once(
529 # "`use_cache=True` is incompatible with gradient checkpointing. Setting `use_cache=False`..."
530 # )
531 use_cache = False
532
533 all_self_attentions = None
534 all_hidden_states = () if output_hidden_states else None
535 for index in range(self.num_layers):
536 if output_hidden_states:
537 all_hidden_states = all_hidden_states + (hidden_states,)
538
539 layer = self._get_layer(index)
540 if self.gradient_checkpointing and self.training:
541 layer_ret = torch.utils.checkpoint.checkpoint(
542 layer,
543 hidden_states,
544 attention_mask,
545 rotary_pos_emb,
546 kv_caches[index],
547 use_cache,
548 use_reentrant=False
549 )
550 else:
551 layer_ret = layer(
552 hidden_states,
553 attention_mask,
554 rotary_pos_emb,
555 kv_cache=kv_caches[index],
556 use_cache=use_cache
557 )
558 hidden_states, kv_cache = layer_ret
559 if use_cache:
560 presents = presents + (kv_cache,)
561
562 if output_hidden_states:
563 all_hidden_states = all_hidden_states + (hidden_states,)
564
565 # Final layer norm.
566 if self.post_layer_norm:
567 hidden_states = self.final_layernorm(hidden_states)
568
569 return hidden_states, presents, all_hidden_states, all_self_attentions
570
571
572class ChatGLMPreTrainedModel(PreTrainedModel):

Callers

nothing calls this directly

Calls 1

_get_layerMethod · 0.95

Tested by

no test coverage detected