CoolFace
Modelpublic

Proteus-Computer-Use/omniparser-finetuned

sourceHugging Facemitupdated 6mo agoView on Hugging Face
4likes49downloads
configuration_florence2.py341 linesDownload Raw Back to root
1# coding=utf-82# Copyright 2024 Microsoft and the HuggingFace Inc. team. All rights reserved.3# Licensed under the Apache License, Version 2.0 (the "License");4# you may not use this file except in compliance with the License.5# You may obtain a copy of the License at6#7#     http://www.apache.org/licenses/LICENSE-2.08#9# Unless required by applicable law or agreed to in writing, software10# distributed under the License is distributed on an "AS IS" BASIS,11# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.12# See the License for the specific language governing permissions and13# limitations under the License.14import warnings15""" Florence-2 configuration"""16 17from typing import Optional18 19from transformers import AutoConfig20from transformers.configuration_utils import PretrainedConfig21from transformers.utils import logging22 23logger = logging.get_logger(__name__)24 25class Florence2VisionConfig(PretrainedConfig):26    r"""27    This is the configuration class to store the configuration of a [`Florence2VisionModel`]. It is used to instantiate a Florence2VisionModel28    according to the specified arguments, defining the model architecture. Instantiating a configuration with the 29    defaults will yield a similar configuration to that of the Florence2VisionModel architecture.30 31    Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the32    documentation from [`PretrainedConfig`] for more information.33 34    Args:35        drop_path_rate (`float`, *optional*, defaults to 0.1):36            The dropout rate of the drop path layer.37        patch_size (`List[int]`, *optional*, defaults to [7, 3, 3, 3]):38            The patch size of the image.39        patch_stride (`List[int]`, *optional*, defaults to [4, 2, 2, 2]):40            The patch stride of the image.41        patch_padding (`List[int]`, *optional*, defaults to [3, 1, 1, 1]):42            The patch padding of the image.43        patch_prenorm (`List[bool]`, *optional*, defaults to [false, true, true, true]):44            Whether to apply layer normalization before the patch embedding layer.45        enable_checkpoint (`bool`, *optional*, defaults to False):46            Whether to enable checkpointing.47        dim_embed (`List[int]`, *optional*, defaults to [256, 512, 1024, 2048]):48            The dimension of the embedding layer.49        num_heads (`List[int]`, *optional*, defaults to [8, 16, 32, 64]):50            The number of attention heads.51        num_groups (`List[int]`, *optional*, defaults to [8, 16, 32, 64]):52            The number of groups.53        depths (`List[int]`, *optional*, defaults to [1, 1, 9, 1]):54            The depth of the model.55        window_size (`int`, *optional*, defaults to 12):56            The window size of the model.57        projection_dim (`int`, *optional*, defaults to 1024):58            The dimension of the projection layer.59        visual_temporal_embedding (`dict`, *optional*):60            The configuration of the visual temporal embedding.61        image_pos_embed (`dict`, *optional*):62            The configuration of the image position embedding.63        image_feature_source (`List[str]`, *optional*, defaults to ["spatial_avg_pool", "temporal_avg_pool"]):64            The source of the image feature.65    Example:66 67    ```python68    >>> from transformers import Florence2VisionConfig, Florence2VisionModel69 70    >>> # Initializing a Florence2 Vision style configuration71    >>> configuration = Florence2VisionConfig()72 73    >>> # Initializing a model (with random weights)74    >>> model = Florence2VisionModel(configuration)75 76    >>> # Accessing the model configuration77    >>> configuration = model.config78    ```"""79 80    model_type = "florence2_vision"81    keys_to_ignore_at_inference = ["past_key_values"]82 83    def __init__(84        self,85        drop_path_rate=0.1,86        patch_size=[7, 3, 3, 3],87        patch_stride=[4, 2, 2, 2],88        patch_padding=[3, 1, 1, 1],89        patch_prenorm=[False, True, True, True],90        enable_checkpoint=False,91        dim_embed=[256, 512, 1024, 2048],92        num_heads=[8, 16, 32, 64],93        num_groups=[8, 16, 32, 64],94        depths=[1, 1, 9, 1],95        window_size=12,96        projection_dim=1024,97        visual_temporal_embedding=None,98        image_pos_embed=None,99        image_feature_source=["spatial_avg_pool", "temporal_avg_pool"],100        **kwargs,101    ):102        self.drop_path_rate = drop_path_rate103        self.patch_size = patch_size104        self.patch_stride = patch_stride105        self.patch_padding = patch_padding106        self.patch_prenorm = patch_prenorm107        self.enable_checkpoint = enable_checkpoint108        self.dim_embed = dim_embed109        self.num_heads = num_heads110        self.num_groups = num_groups111        self.depths = depths112        self.window_size = window_size113        self.projection_dim = projection_dim114        self.visual_temporal_embedding = visual_temporal_embedding115        self.image_pos_embed = image_pos_embed116        self.image_feature_source = image_feature_source117 118        super().__init__(**kwargs)119 120 121 122class Florence2LanguageConfig(PretrainedConfig):123    r"""124    This is the configuration class to store the configuration of a [`Florence2LanguagePreTrainedModel`]. It is used to instantiate a BART125    model according to the specified arguments, defining the model architecture. Instantiating a configuration with the126    defaults will yield a similar configuration to that of the BART127    [facebook/bart-large](https://huggingface.co/facebook/bart-large) architecture.128 129    Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the130    documentation from [`PretrainedConfig`] for more information.131 132 133    Args:134        vocab_size (`int`, *optional*, defaults to 51289):135            Vocabulary size of the Florence2Language model. Defines the number of different tokens that can be represented by the136            `inputs_ids` passed when calling [`Florence2LanguageModel`].137        d_model (`int`, *optional*, defaults to 1024):138            Dimensionality of the layers and the pooler layer.139        encoder_layers (`int`, *optional*, defaults to 12):140            Number of encoder layers.141        decoder_layers (`int`, *optional*, defaults to 12):142            Number of decoder layers.143        encoder_attention_heads (`int`, *optional*, defaults to 16):144            Number of attention heads for each attention layer in the Transformer encoder.145        decoder_attention_heads (`int`, *optional*, defaults to 16):146            Number of attention heads for each attention layer in the Transformer decoder.147        decoder_ffn_dim (`int`, *optional*, defaults to 4096):148            Dimensionality of the "intermediate" (often named feed-forward) layer in decoder.149        encoder_ffn_dim (`int`, *optional*, defaults to 4096):150            Dimensionality of the "intermediate" (often named feed-forward) layer in decoder.151        activation_function (`str` or `function`, *optional*, defaults to `"gelu"`):152            The non-linear activation function (function or string) in the encoder and pooler. If string, `"gelu"`,153            `"relu"`, `"silu"` and `"gelu_new"` are supported.154        dropout (`float`, *optional*, defaults to 0.1):155            The dropout probability for all fully connected layers in the embeddings, encoder, and pooler.156        attention_dropout (`float`, *optional*, defaults to 0.0):157            The dropout ratio for the attention probabilities.158        activation_dropout (`float`, *optional*, defaults to 0.0):159            The dropout ratio for activations inside the fully connected layer.160        classifier_dropout (`float`, *optional*, defaults to 0.0):161            The dropout ratio for classifier.162        max_position_embeddings (`int`, *optional*, defaults to 1024):163            The maximum sequence length that this model might ever be used with. Typically set this to something large164            just in case (e.g., 512 or 1024 or 2048).165        init_std (`float`, *optional*, defaults to 0.02):166            The standard deviation of the truncated_normal_initializer for initializing all weight matrices.167        encoder_layerdrop (`float`, *optional*, defaults to 0.0):168            The LayerDrop probability for the encoder. See the [LayerDrop paper](see https://arxiv.org/abs/1909.11556)169            for more details.170        decoder_layerdrop (`float`, *optional*, defaults to 0.0):171            The LayerDrop probability for the decoder. See the [LayerDrop paper](see https://arxiv.org/abs/1909.11556)172            for more details.173        scale_embedding (`bool`, *optional*, defaults to `False`):174            Scale embeddings by diving by sqrt(d_model).175        use_cache (`bool`, *optional*, defaults to `True`):176            Whether or not the model should return the last key/values attentions (not used by all models).177        num_labels (`int`, *optional*, defaults to 3):178            The number of labels to use in [`Florence2LanguageForSequenceClassification`].179        forced_eos_token_id (`int`, *optional*, defaults to 2):180            The id of the token to force as the last generated token when `max_length` is reached. Usually set to181            `eos_token_id`.182 183    Example:184 185    ```python186    >>> from transformers import Florence2LanguageConfig, Florence2LanguageModel187 188    >>> # Initializing a Florence2 Language style configuration189    >>> configuration = Florence2LanguageConfig()190 191    >>> # Initializing a model (with random weights)192    >>> model = Florence2LangaugeModel(configuration)193 194    >>> # Accessing the model configuration195    >>> configuration = model.config196    ```"""197 198    model_type = "florence2_language"199    keys_to_ignore_at_inference = ["past_key_values"]200    attribute_map = {"num_attention_heads": "encoder_attention_heads", "hidden_size": "d_model"}201 202    def __init__(203        self,204        vocab_size=51289,205        max_position_embeddings=1024,206        encoder_layers=12,207        encoder_ffn_dim=4096,208        encoder_attention_heads=16,209        decoder_layers=12,210        decoder_ffn_dim=4096,211        decoder_attention_heads=16,212        encoder_layerdrop=0.0,213        decoder_layerdrop=0.0,214        activation_function="gelu",215        d_model=1024,216        dropout=0.1,217        attention_dropout=0.0,218        activation_dropout=0.0,219        init_std=0.02,220        classifier_dropout=0.0,221        scale_embedding=False,222        use_cache=True,223        num_labels=3,224        pad_token_id=1,225        bos_token_id=0,226        eos_token_id=2,227        is_encoder_decoder=True,228        decoder_start_token_id=2,229        forced_eos_token_id=2,230        **kwargs,231    ):232        self.vocab_size = vocab_size233        self.max_position_embeddings = max_position_embeddings234        self.d_model = d_model235        self.encoder_ffn_dim = encoder_ffn_dim236        self.encoder_layers = encoder_layers237        self.encoder_attention_heads = encoder_attention_heads238        self.decoder_ffn_dim = decoder_ffn_dim239        self.decoder_layers = decoder_layers240        self.decoder_attention_heads = decoder_attention_heads241        self.dropout = dropout242        self.attention_dropout = attention_dropout243        self.activation_dropout = activation_dropout244        self.activation_function = activation_function245        self.init_std = init_std246        self.encoder_layerdrop = encoder_layerdrop247        self.decoder_layerdrop = decoder_layerdrop248        self.classifier_dropout = classifier_dropout249        self.use_cache = use_cache250        self.num_hidden_layers = encoder_layers251        self.scale_embedding = scale_embedding  # scale factor will be sqrt(d_model) if True252 253        super().__init__(254            num_labels=num_labels,255            pad_token_id=pad_token_id,256            bos_token_id=bos_token_id,257            eos_token_id=eos_token_id,258            is_encoder_decoder=is_encoder_decoder,259            decoder_start_token_id=decoder_start_token_id,260            forced_eos_token_id=forced_eos_token_id,261            **kwargs,262        )263 264        # ensure backward compatibility for BART CNN models265        if self.forced_bos_token_id is None and kwargs.get("force_bos_token_to_be_generated", False):266            self.forced_bos_token_id = self.bos_token_id267            warnings.warn(268                f"Please make sure the config includes `forced_bos_token_id={self.bos_token_id}` in future versions. "269                "The config can simply be saved and uploaded again to be fixed."270            )271 272class Florence2Config(PretrainedConfig):273    r"""274    This is the configuration class to store the configuration of a [`Florence2ForConditionalGeneration`]. It is used to instantiate an275    Florence-2 model according to the specified arguments, defining the model architecture. 276 277    Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the278    documentation from [`PretrainedConfig`] for more information.279 280    Args:281        vision_config (`Florence2VisionConfig`,  *optional*):282            Custom vision config or dict283        text_config (`Union[AutoConfig, dict]`, *optional*):284            The config object of the text backbone. 285        ignore_index (`int`, *optional*, defaults to -100):286            The ignore index for the loss function.287        vocab_size (`int`, *optional*, defaults to 51289):288            Vocabulary size of the Florence2model. Defines the number of different tokens that can be represented by the289            `inputs_ids` passed when calling [`~Florence2ForConditionalGeneration`]290        projection_dim (`int`, *optional*, defaults to 1024):291            Dimension of the multimodal projection space.292 293    Example:294 295    ```python296    >>> from transformers import Florence2ForConditionalGeneration, Florence2Config, CLIPVisionConfig, BartConfig297 298    >>> # Initializing a clip-like vision config299    >>> vision_config = CLIPVisionConfig()300 301    >>> # Initializing a Bart config302    >>> text_config = BartConfig()303 304    >>> # Initializing a Florence-2 configuration305    >>> configuration = Florence2Config(vision_config, text_config)306 307    >>> # Initializing a model from the florence-2 configuration308    >>> model = Florence2ForConditionalGeneration(configuration)309 310    >>> # Accessing the model configuration311    >>> configuration = model.config312    ```"""313 314    model_type = "florence2"315    is_composition = False316 317    def __init__(318        self,319        vision_config=None,320        text_config=None,321        ignore_index=-100,322        vocab_size=51289,323        projection_dim=1024,324        **kwargs,325    ):326        self.ignore_index = ignore_index327        self.vocab_size = vocab_size328        self.projection_dim = projection_dim329        if vision_config is not None:330            vision_config = PretrainedConfig(**vision_config)331        self.vision_config = vision_config332        self.vocab_size = self.vocab_size333 334        self.text_config = text_config335        if text_config is not None:336            self.text_config = Florence2LanguageConfig(**text_config)337 338 339        super().__init__(**kwargs)340 341