ByteDance/Sa2VA-InternVL3-2B
2190
1# Copyright 2024 Microsoft and the HuggingFace Inc. team. All rights reserved.2#3# Licensed under the Apache License, Version 2.0 (the "License");4# you may not use this file except in compliance with the License.5# You may obtain a copy of the License atd6#7# http://www.apache.org/licenses/LICENSE-2.08#9# Unless required by applicable law or agreed to in writing, software10# distributed under the License is distributed on an "AS IS" BASIS,11# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.12# See the License for the specific language governing permissions and13# limitations under the License.14 15""" Phi-3 model configuration"""16 17 18from transformers.configuration_utils import PretrainedConfig19from transformers.utils import logging20 21logger = logging.get_logger(__name__)22 23PHI3_PRETRAINED_CONFIG_ARCHIVE_MAP = {24 'microsoft/Phi-3-mini-4k-instruct': 'https://huggingface.co/microsoft/Phi-3-mini-4k-instruct/resolve/main/config.json',25 'microsoft/Phi-3-mini-128k-instruct': 'https://huggingface.co/microsoft/Phi-3-mini-128k-instruct/resolve/main/config.json',26}27 28 29class Phi3Config(PretrainedConfig):30 r"""31 This is the configuration class to store the configuration of a [`Phi3Model`]. It is used to instantiate a Phi-332 model according to the specified arguments, defining the model architecture. Instantiating a configuration with the33 defaults will yield a similar configuration to that of the34 [microsoft/Phi-3-mini-4k-instruct](https://huggingface.co/microsoft/Phi-3-mini-4k-instruct).35 36 Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the37 documentation from [`PretrainedConfig`] for more information.38 39 Args:40 vocab_size (`int`, *optional*, defaults to 32064):41 Vocabulary size of the Phi-3 model. Defines the number of different tokens that can be represented by the42 `inputs_ids` passed when calling [`Phi3Model`].43 hidden_size (`int`, *optional*, defaults to 3072):44 Dimension of the hidden representations.45 intermediate_size (`int`, *optional*, defaults to 8192):46 Dimension of the MLP representations.47 num_hidden_layers (`int`, *optional*, defaults to 32):48 Number of hidden layers in the Transformer decoder.49 num_attention_heads (`int`, *optional*, defaults to 32):50 Number of attention heads for each attention layer in the Transformer decoder.51 num_key_value_heads (`int`, *optional*):52 This is the number of key_value heads that should be used to implement Grouped Query Attention. If53 `num_key_value_heads=num_attention_heads`, the model will use Multi Head Attention (MHA), if54 `num_key_value_heads=1 the model will use Multi Query Attention (MQA) otherwise GQA is used. When55 converting a multi-head checkpoint to a GQA checkpoint, each group key and value head should be constructed56 by meanpooling all the original heads within that group. For more details checkout [this57 paper](https://arxiv.org/pdf/2305.13245.pdf). If it is not specified, will default to58 `num_attention_heads`.59 resid_pdrop (`float`, *optional*, defaults to 0.0):60 Dropout probability for mlp outputs.61 embd_pdrop (`int`, *optional*, defaults to 0.0):62 The dropout ratio for the embeddings.63 attention_dropout (`float`, *optional*, defaults to 0.0):64 The dropout ratio after computing the attention scores.65 hidden_act (`str` or `function`, *optional*, defaults to `"silu"`):66 The non-linear activation function (function or string) in the decoder.67 max_position_embeddings (`int`, *optional*, defaults to 4096):68 The maximum sequence length that this model might ever be used with.69 original_max_position_embeddings (`int`, *optional*, defaults to 4096):70 The maximum sequence length that this model was trained with. This is used to determine the size of the71 original RoPE embeddings when using long scaling.72 initializer_range (`float`, *optional*, defaults to 0.02):73 The standard deviation of the truncated_normal_initializer for initializing all weight matrices.74 rms_norm_eps (`float`, *optional*, defaults to 1e-05):75 The epsilon value used for the RMSNorm.76 use_cache (`bool`, *optional*, defaults to `True`):77 Whether or not the model should return the last key/values attentions (not used by all models). Only78 relevant if `config.is_decoder=True`. Whether to tie weight embeddings or not.79 tie_word_embeddings (`bool`, *optional*, defaults to `False`):80 Whether to tie weight embeddings81 rope_theta (`float`, *optional*, defaults to 10000.0):82 The base period of the RoPE embeddings.83 rope_scaling (`dict`, *optional*):84 The scaling strategy for the RoPE embeddings. If `None`, no scaling is applied. If a dictionary, it must85 contain the following keys: `type`, `short_factor` and `long_factor`. The `type` must be either `su` or `yarn` and86 the `short_factor` and `long_factor` must be lists of numbers with the same length as the hidden size87 divided by the number of attention heads divided by 2.88 bos_token_id (`int`, *optional*, defaults to 1):89 The id of the "beginning-of-sequence" token.90 eos_token_id (`int`, *optional*, defaults to 32000):91 The id of the "end-of-sequence" token.92 pad_token_id (`int`, *optional*, defaults to 32000):93 The id of the padding token.94 sliding_window (`int`, *optional*):95 Sliding window attention window size. If `None`, no sliding window is applied.96 97 Example:98 99 ```python100 >>> from transformers import Phi3Model, Phi3Config101 102 >>> # Initializing a Phi-3 style configuration103 >>> configuration = Phi3Config.from_pretrained("microsoft/Phi-3-mini-4k-instruct")104 105 >>> # Initializing a model from the configuration106 >>> model = Phi3Model(configuration)107 108 >>> # Accessing the model configuration109 >>> configuration = model.config110 ```"""111 112 model_type = 'phi3'113 keys_to_ignore_at_inference = ['past_key_values']114 115 def __init__(116 self,117 vocab_size=32064,118 hidden_size=3072,119 intermediate_size=8192,120 num_hidden_layers=32,121 num_attention_heads=32,122 num_key_value_heads=None,123 resid_pdrop=0.0,124 embd_pdrop=0.0,125 attention_dropout=0.0,126 hidden_act='silu',127 max_position_embeddings=4096,128 original_max_position_embeddings=4096,129 initializer_range=0.02,130 rms_norm_eps=1e-5,131 use_cache=True,132 tie_word_embeddings=False,133 rope_theta=10000.0,134 rope_scaling=None,135 bos_token_id=1,136 eos_token_id=32000,137 pad_token_id=32000,138 sliding_window=None,139 **kwargs,140 ):141 self.vocab_size = vocab_size142 self.hidden_size = hidden_size143 self.intermediate_size = intermediate_size144 self.num_hidden_layers = num_hidden_layers145 self.num_attention_heads = num_attention_heads146 147 if num_key_value_heads is None:148 num_key_value_heads = num_attention_heads149 150 self.num_key_value_heads = num_key_value_heads151 self.resid_pdrop = resid_pdrop152 self.embd_pdrop = embd_pdrop153 self.attention_dropout = attention_dropout154 self.hidden_act = hidden_act155 self.max_position_embeddings = max_position_embeddings156 self.original_max_position_embeddings = original_max_position_embeddings157 self.initializer_range = initializer_range158 self.rms_norm_eps = rms_norm_eps159 self.use_cache = use_cache160 self.rope_theta = rope_theta161 self.rope_scaling = rope_scaling162 self._rope_scaling_validation()163 self.sliding_window = sliding_window164 165 super().__init__(166 bos_token_id=bos_token_id,167 eos_token_id=eos_token_id,168 pad_token_id=pad_token_id,169 tie_word_embeddings=tie_word_embeddings,170 **kwargs,171 )172 173 def _rope_scaling_validation(self):174 """175 Validate the `rope_scaling` configuration.176 """177 if self.rope_scaling is None:178 return179 180 if not isinstance(self.rope_scaling, dict) or len(self.rope_scaling) != 3:181 raise ValueError(182 '`rope_scaling` must be a dictionary with three fields, `type`, `short_factor` and `long_factor`, '183 f'got {self.rope_scaling}'184 )185 rope_scaling_type = self.rope_scaling.get('type', None)186 rope_scaling_short_factor = self.rope_scaling.get('short_factor', None)187 rope_scaling_long_factor = self.rope_scaling.get('long_factor', None)188 if rope_scaling_type is None or rope_scaling_type not in ['su', 'yarn']:189 raise ValueError(f"`rope_scaling`'s type field must be one of ['su', 'yarn'], got {rope_scaling_type}")190 if not (191 isinstance(rope_scaling_short_factor, list)192 and all(isinstance(x, (int, float)) for x in rope_scaling_short_factor)193 ):194 raise ValueError(195 f"`rope_scaling`'s short_factor field must be a list of numbers, got {rope_scaling_short_factor}"196 )197 if not len(rope_scaling_short_factor) == self.hidden_size // self.num_attention_heads // 2:198 raise ValueError(199 f"`rope_scaling`'s short_factor field must have length {self.hidden_size // self.num_attention_heads // 2}, got {len(rope_scaling_short_factor)}"200 )201 if not (202 isinstance(rope_scaling_long_factor, list)203 and all(isinstance(x, (int, float)) for x in rope_scaling_long_factor)204 ):205 raise ValueError(206 f"`rope_scaling`'s long_factor field must be a list of numbers, got {rope_scaling_long_factor}"207 )208 if not len(rope_scaling_long_factor) == self.hidden_size // self.num_attention_heads // 2:209 raise ValueError(210 f"`rope_scaling`'s long_factor field must have length {self.hidden_size // self.num_attention_heads // 2}, got {len(rope_scaling_long_factor)}"211 )212 