Upload modeling_internvl_chat.py with huggingface_hub
Browse files
modeling_internvl_chat.py
CHANGED
|
@@ -12,7 +12,7 @@ import transformers
|
|
| 12 |
from torch import nn
|
| 13 |
from torch.nn import CrossEntropyLoss
|
| 14 |
from transformers import (AutoModel, GenerationConfig, LlamaForCausalLM,
|
| 15 |
-
|
| 16 |
from transformers.modeling_outputs import CausalLMOutputWithPast
|
| 17 |
from transformers.modeling_utils import PreTrainedModel
|
| 18 |
from transformers.utils import ModelOutput, logging
|
|
@@ -20,7 +20,6 @@ from transformers.utils import ModelOutput, logging
|
|
| 20 |
from .configuration_internvl_chat import InternVLChatConfig
|
| 21 |
from .conversation import get_conv_template
|
| 22 |
from .modeling_intern_vit import InternVisionModel, has_flash_attn
|
| 23 |
-
from .modeling_internlm2 import InternLM2ForCausalLM
|
| 24 |
|
| 25 |
logger = logging.get_logger(__name__)
|
| 26 |
|
|
@@ -39,7 +38,7 @@ class InternVLChatModel(PreTrainedModel):
|
|
| 39 |
base_model_prefix = 'language_model'
|
| 40 |
_supports_flash_attn_2 = True
|
| 41 |
supports_gradient_checkpointing = True
|
| 42 |
-
_no_split_modules = ['InternVisionModel', 'LlamaDecoderLayer', '
|
| 43 |
|
| 44 |
def __init__(self, config: InternVLChatConfig, vision_model=None, language_model=None, use_flash_attn=True):
|
| 45 |
super().__init__(config)
|
|
@@ -55,7 +54,7 @@ class InternVLChatModel(PreTrainedModel):
|
|
| 55 |
self.ps_version = config.ps_version
|
| 56 |
use_flash_attn = use_flash_attn if has_flash_attn else False
|
| 57 |
config.vision_config.use_flash_attn = True if use_flash_attn else False
|
| 58 |
-
config.llm_config.
|
| 59 |
|
| 60 |
logger.info(f'num_image_token: {self.num_image_token}')
|
| 61 |
logger.info(f'ps_version: {self.ps_version}')
|
|
@@ -68,8 +67,8 @@ class InternVLChatModel(PreTrainedModel):
|
|
| 68 |
else:
|
| 69 |
if config.llm_config.architectures[0] == 'LlamaForCausalLM':
|
| 70 |
self.language_model = LlamaForCausalLM(config.llm_config)
|
| 71 |
-
elif config.llm_config.architectures[0] == '
|
| 72 |
-
self.language_model =
|
| 73 |
else:
|
| 74 |
raise NotImplementedError(f'{config.llm_config.architectures[0]} is not implemented.')
|
| 75 |
|
|
|
|
| 12 |
from torch import nn
|
| 13 |
from torch.nn import CrossEntropyLoss
|
| 14 |
from transformers import (AutoModel, GenerationConfig, LlamaForCausalLM,
|
| 15 |
+
Qwen2ForCausalLM)
|
| 16 |
from transformers.modeling_outputs import CausalLMOutputWithPast
|
| 17 |
from transformers.modeling_utils import PreTrainedModel
|
| 18 |
from transformers.utils import ModelOutput, logging
|
|
|
|
| 20 |
from .configuration_internvl_chat import InternVLChatConfig
|
| 21 |
from .conversation import get_conv_template
|
| 22 |
from .modeling_intern_vit import InternVisionModel, has_flash_attn
|
|
|
|
| 23 |
|
| 24 |
logger = logging.get_logger(__name__)
|
| 25 |
|
|
|
|
| 38 |
base_model_prefix = 'language_model'
|
| 39 |
_supports_flash_attn_2 = True
|
| 40 |
supports_gradient_checkpointing = True
|
| 41 |
+
_no_split_modules = ['InternVisionModel', 'LlamaDecoderLayer', 'Qwen2DecoderLayer']
|
| 42 |
|
| 43 |
def __init__(self, config: InternVLChatConfig, vision_model=None, language_model=None, use_flash_attn=True):
|
| 44 |
super().__init__(config)
|
|
|
|
| 54 |
self.ps_version = config.ps_version
|
| 55 |
use_flash_attn = use_flash_attn if has_flash_attn else False
|
| 56 |
config.vision_config.use_flash_attn = True if use_flash_attn else False
|
| 57 |
+
config.llm_config._attn_implementation = 'flash_attention_2' if use_flash_attn else 'eager'
|
| 58 |
|
| 59 |
logger.info(f'num_image_token: {self.num_image_token}')
|
| 60 |
logger.info(f'ps_version: {self.ps_version}')
|
|
|
|
| 67 |
else:
|
| 68 |
if config.llm_config.architectures[0] == 'LlamaForCausalLM':
|
| 69 |
self.language_model = LlamaForCausalLM(config.llm_config)
|
| 70 |
+
elif config.llm_config.architectures[0] == 'Qwen2ForCausalLM':
|
| 71 |
+
self.language_model = Qwen2ForCausalLM(config.llm_config)
|
| 72 |
else:
|
| 73 |
raise NotImplementedError(f'{config.llm_config.architectures[0]} is not implemented.')
|
| 74 |
|