Replace _flash_attention_forward to _ulysses_flash_attention_forward
(model: PreTrainedModel)
| 85 | |
| 86 | |
| 87 | def apply_monkey_patch(model: PreTrainedModel): |
| 88 | """Replace _flash_attention_forward to _ulysses_flash_attention_forward""" |
| 89 | module = sys.modules[model.__module__] |
| 90 | |
| 91 | # TODO: VLM models only, unify monkey patch to LLM models. |
| 92 | if model.config.model_type in ("qwen2_vl", "qwen2_5_vl"): # patch remove padding for qwen2vl mrope |
| 93 | from verl.models.transformers.qwen2_vl import ulysses_flash_attn_forward |
| 94 | from transformers.models.qwen2_vl.modeling_qwen2_vl import Qwen2VLFlashAttention2 |
| 95 | from transformers.models.qwen2_5_vl.modeling_qwen2_5_vl import Qwen2_5_VLFlashAttention2 |
| 96 | |
| 97 | Qwen2VLFlashAttention2.forward = ulysses_flash_attn_forward |
| 98 | Qwen2_5_VLFlashAttention2.forward = ulysses_flash_attn_forward |
| 99 | print("Monkey patch FlashAttention2.forward in Qwen2VL") |
| 100 | return |
| 101 | |
| 102 | # transformers<=4.47.1 |
| 103 | if hasattr(module, "_flash_attention_forward"): |
| 104 | module._flash_attention_forward = _ulysses_flash_attention_forward |
| 105 | print(f"Monkey patch _flash_attention_forward in {model.__module__}") |
| 106 | else: |
| 107 | # transformers>=4.48.0 |
| 108 | from transformers.integrations import flash_attention |
| 109 | flash_attention._flash_attention_forward = _ulysses_flash_attention_forward |
| 110 | print(f"Monkey patch _flash_attention_forward in {flash_attention.__name__}") |
| 111 | |
| 112 | |
| 113 | from functools import lru_cache |
no outgoing calls
no test coverage detected