Ë
    T^(h»=  ã                   óŽ   — d Z ddlmZ ddlmZ  ej
                  e«      Z G d„ de«      Z G d„ de«      Z	 G d„ d	e«      Z
g d
¢Zy)zPix2Struct model configurationé   )ÚPretrainedConfig)Úloggingc                   óf   ‡ — e Zd ZdZdZdgZddddddddœZ	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 d	ˆ fd„	Zˆ xZS )
ÚPix2StructTextConfigaç  
    This is the configuration class to store the configuration of a [`Pix2StructTextModel`]. It is used to instantiate
    a Pix2Struct text model according to the specified arguments, defining the model architecture. Instantiating a
    configuration with the defaults will yield a similar configuration to that of the Pix2Struct text decoder used by
    the [google/pix2struct-base](https://huggingface.co/google/pix2struct-base) architecture.

    Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
    documentation from [`PretrainedConfig`] for more information.

    Args:
        vocab_size (`int`, *optional*, defaults to 50244):
            Vocabulary size of the `Pix2Struct` text model. Defines the number of different tokens that can be
            represented by the `inputs_ids` passed when calling [`Pix2StructTextModel`].
        hidden_size (`int`, *optional*, defaults to 768):
            Dimensionality of the encoder layers and the pooler layer.
        d_kv (`int`, *optional*, defaults to 64):
            Dimensionality of the key, query, value projections in each attention head.
        d_ff (`int`, *optional*, defaults to 2048):
            Dimensionality of the "intermediate" (i.e., feed-forward) layer in the Transformer encoder.
        num_layers (`int`, *optional*, defaults to 12):
            Number of hidden layers in the Transformer encoder.
        num_heads (`int`, *optional*, defaults to 12):
            Number of attention heads for each attention layer in the Transformer encoder.
        relative_attention_num_buckets (`int`, *optional*, defaults to 32):
            The number of buckets to use for each attention layer.
        relative_attention_max_distance (`int`, *optional*, defaults to 128):
            The maximum distance of the longer sequences for the bucket separation.
        dropout_rate (`float`, *optional*, defaults to 0.1):
            The dropout probability for all fully connected layers in the embeddings, encoder, and pooler.
        layer_norm_epsilon (`float`, *optional*, defaults to 1e-6):
            The epsilon used by the layer normalization layers.
        initializer_factor (`float`, *optional*, defaults to 1.0):
            A factor for initializing all weight matrices (should be kept to 1, used internally for initialization
            testing).
        dense_act_fn (`Union[Callable, str]`, *optional*, defaults to `"gelu_new"`):
            The non-linear activation function (function or string).
        decoder_start_token_id (`int`, *optional*, defaults to 0):
            The id of the `decoder_start_token_id` token.
        use_cache (`bool`, *optional*, defaults to `False`):
            Whether or not the model should return the last key/values attentions (not used by all models).
        pad_token_id (`int`, *optional*, defaults to 0):
            The id of the `padding` token.
        eos_token_id (`int`, *optional*, defaults to 1):
            The id of the `end-of-sequence` token.

    Example:

    ```python
    >>> from transformers import Pix2StructTextConfig, Pix2StructTextModel

    >>> # Initializing a Pix2StructTextConfig with google/pix2struct-base style configuration
    >>> configuration = Pix2StructTextConfig()

    >>> # Initializing a Pix2StructTextModel (with random weights) from the google/pix2struct-base style configuration
    >>> model = Pix2StructTextModel(configuration)

    >>> # Accessing the model configuration
    >>> configuration = model.config
    ```Úpix2struct_text_modelÚpast_key_valuesÚhidden_sizeÚ	num_headsÚ
num_layers)r	   Únum_attention_headsÚnum_hidden_layersÚdecoder_attention_headsÚencoder_attention_headsÚencoder_layersÚdecoder_layersc           	      ó  •— || _         || _        || _        || _        || _        || _        || _        || _        |	| _        |
| _	        || _
        || _        || _        || _        || _        t        ‰| �@  d|||||dœ|¤Ž y )N)Úpad_token_idÚeos_token_idÚdecoder_start_token_idÚtie_word_embeddingsÚ
is_decoder© )Ú
vocab_sizer	   Úd_kvÚd_ffr   r
   Úrelative_attention_num_bucketsÚrelative_attention_max_distanceÚdropout_rateÚlayer_norm_epsilonÚinitializer_factorÚ	use_cacher   r   Údense_act_fnÚsuperÚ__init__)Úselfr   r	   r   r   r   r
   r   r   r   r   r    r"   r   r!   r   r   r   r   ÚkwargsÚ	__class__s                       €úu/var/www/skyplay_api_hub/venv/lib/python3.12/site-packages/transformers/models/pix2struct/configuration_pix2struct.pyr$   zPix2StructTextConfig.__init__a   s¤   ø€ ð, %ˆŒØ&ˆÔØˆŒ	ØˆŒ	Ø$ˆŒØ"ˆŒØ.LˆÔ+Ø/NˆÔ,Ø(ˆÔØ"4ˆÔØ"4ˆÔØ"ˆŒà(ˆÔØ&<ˆÔ#ð )ˆÔä‰Ñð 	
Ø%Ø%Ø#9Ø 3Ø!ñ	
ð ó	
ó    )iDÄ  é   é@   é   é   r-   é    é€   gš™™™™™¹?ç�íµ ÷Æ°>ç      ð?Úgelu_newé    Fr3   é   FT)	Ú__name__Ú
__module__Ú__qualname__Ú__doc__Ú
model_typeÚkeys_to_ignore_at_inferenceÚattribute_mapr$   Ú__classcell__©r'   s   @r(   r   r      sw   ø„ ñ:ðx )€JØ#4Ð"5Ðà$Ø*Ø)Ø#.Ø#.Ø&Ø&ñ€Mð ØØØØØØ')Ø(+ØØØØØ ØØØØ!Ø÷'0
ñ 0
r)   r   c                   óF   ‡ — e Zd ZdZdZ	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 dˆ fd„	Zˆ xZS )ÚPix2StructVisionConfiga·  
    This is the configuration class to store the configuration of a [`Pix2StructVisionModel`]. It is used to
    instantiate a Pix2Struct vision model according to the specified arguments, defining the model architecture.
    Instantiating a configuration defaults will yield a similar configuration to that of the Pix2Struct-base
    [google/pix2struct-base](https://huggingface.co/google/pix2struct-base) architecture.

    Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
    documentation from [`PretrainedConfig`] for more information.

    Args:
        hidden_size (`int`, *optional*, defaults to 768):
            Dimensionality of the encoder layers and the pooler layer.
        patch_embed_hidden_size (`int`, *optional*, defaults to 768):
            Dimensionality of the input patch_embedding layer in the Transformer encoder.
        d_ff (`int`, *optional*, defaults to 2048):
            Dimensionality of the "intermediate" (i.e., feed-forward) layer in the Transformer encoder.
        d_kv (`int`, *optional*, defaults to 64):
            Dimensionality of the key, query, value projections per attention head.
        num_hidden_layers (`int`, *optional*, defaults to 12):
            Number of hidden layers in the Transformer encoder.
        num_attention_heads (`int`, *optional*, defaults to 12):
            Number of attention heads for each attention layer in the Transformer encoder.
        dense_act_fn (`str` or `function`, *optional*, defaults to `"gelu_new"`):
            The non-linear activation function (function or string) in the encoder and pooler. If string, `"gelu"`,
            `"relu"`, `"selu"` and `"gelu_new"` `"gelu"` are supported.
        layer_norm_eps (`float`, *optional*, defaults to 1e-06):
            The epsilon used by the layer normalization layers.
        dropout_rate (`float`, *optional*, defaults to 0.0):
            The dropout probability for all fully connected layers in the embeddings, encoder, and pooler.
        attention_dropout (`float`, *optional*, defaults to 0.0):
            The dropout ratio for the attention probabilities.
        initializer_range (`float`, *optional*, defaults to 1e-10):
            The standard deviation of the truncated_normal_initializer for initializing all weight matrices.
        initializer_factor (`float`, *optional*, defaults to 1.0):
            A factor for initializing all weight matrices (should be kept to 1, used internally for initialization
            testing).
        seq_len (`int`, *optional*, defaults to 4096):
            Maximum sequence length (here number of patches) supported by the model.
        relative_attention_num_buckets (`int`, *optional*, defaults to 32):
            The number of buckets to use for each attention layer.
        relative_attention_max_distance (`int`, *optional*, defaults to 128):
            The maximum distance (in tokens) to use for each attention layer.

    Example:

    ```python
    >>> from transformers import Pix2StructVisionConfig, Pix2StructVisionModel

    >>> # Initializing a Pix2StructVisionConfig with google/pix2struct-base style configuration
    >>> configuration = Pix2StructVisionConfig()

    >>> # Initializing a Pix2StructVisionModel (with random weights) from the google/pix2struct-base style configuration
    >>> model = Pix2StructVisionModel(configuration)

    >>> # Accessing the model configuration
    >>> configuration = model.config
    ```Úpix2struct_vision_modelc                 óö   •— t        ‰| �  di |¤Ž || _        || _        || _        |	| _        || _        || _        || _        || _	        |
| _
        || _        || _        || _        || _        || _        || _        y )Nr   )r#   r$   r	   Úpatch_embed_hidden_sizer   r   r   r   Úinitializer_ranger    Úattention_dropoutÚlayer_norm_epsr"   Úseq_lenr   r   r   )r%   r	   rB   r   r   r   r   r"   rE   r   rD   rC   r    rF   r   r   r&   r'   s                    €r(   r$   zPix2StructVisionConfig.__init__Ñ   sŠ   ø€ ô& 	‰ÑÑ"˜6Ò"à&ˆÔØ'>ˆÔ$ØˆŒ	Ø(ˆÔØ!2ˆÔØ#6ˆÔ Ø!2ˆÔØ"4ˆÔØ!2ˆÔØ,ˆÔØ(ˆÔØˆŒØ.LˆÔ+Ø/NˆÔ,Øˆ�	r)   )r*   r*   r,   r+   r-   r-   r2   r0   ç        rG   g»½×Ùß|Û=r1   i   r.   r/   )r5   r6   r7   r8   r9   r$   r<   r=   s   @r(   r?   r?   ”   sI   ø„ ñ8ðt +€Jð Ø #ØØØØØØØØØØØØ')Ø(+÷!#ñ #r)   r?   c                   óP   ‡ — e Zd ZdZdZ	 	 	 	 	 	 	 dˆ fd„	Zededefd„«       Z	ˆ xZ
S )ÚPix2StructConfiga1	  
    [`Pix2StructConfig`] is the configuration class to store the configuration of a
    [`Pix2StructForConditionalGeneration`]. It is used to instantiate a Pix2Struct model according to the specified
    arguments, defining the text model and vision model configs. Instantiating a configuration with the defaults will
    yield a similar configuration to that of the Pix2Struct-base
    [google/pix2struct-base](https://huggingface.co/google/pix2struct-base) architecture.

    Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
    documentation from [`PretrainedConfig`] for more information.

    Args:
        text_config (`dict`, *optional*):
            Dictionary of configuration options used to initialize [`Pix2StructTextConfig`].
        vision_config (`dict`, *optional*):
            Dictionary of configuration options used to initialize [`Pix2StructVisionConfig`].
        initializer_factor (`float`, *optional*, defaults to 1.0):
            Factor to multiply the initialization range with.
        initializer_range (`float`, *optional*, defaults to 0.02):
            The standard deviation of the truncated_normal_initializer for initializing all weight matrices.
        is_vqa (`bool`, *optional*, defaults to `False`):
            Whether the model has been fine-tuned for VQA or not.
        kwargs (*optional*):
            Dictionary of keyword arguments.

    Example:

    ```python
    >>> from transformers import Pix2StructConfig, Pix2StructForConditionalGeneration

    >>> # Initializing a Pix2StructConfig with google/pix2struct-base style configuration
    >>> configuration = Pix2StructConfig()

    >>> # Initializing a Pix2StructForConditionalGeneration (with random weights) from the google/pix2struct-base style configuration
    >>> model = Pix2StructForConditionalGeneration(configuration)

    >>> # Accessing the model configuration
    >>> configuration = model.config

    >>> # We can also initialize a Pix2StructConfig from a Pix2StructTextConfig and a Pix2StructVisionConfig

    >>> # Initializing a Pix2Struct text and Pix2Struct vision configuration
    >>> config_text = Pix2StructTextConfig()
    >>> config_vision = Pix2StructVisionConfig()

    >>> config = Pix2StructConfig.from_text_vision_configs(config_text, config_vision)
    ```Ú
pix2structc                 ó  •— t        ‰	| �  d||dœ|¤Ž |€i }t        j                  d«       |€i }t        j                  d«       ||d<   ||d<   t	        di |¤Ž| _        t        di |¤Ž| _        | j
                  j                  | _        | j
                  j                  | _	        | j
                  j                  | _
        || _        || _        | j                  | j
                  _        | j                  | j                  _        || _        y )N)r   Úis_encoder_decoderzOtext_config is None. Initializing the Pix2StructTextConfig with default values.zSvision_config is None. Initializing the Pix2StructVisionConfig with default values.rL   r   r   )r#   r$   ÚloggerÚinfor   Útext_configr?   Úvision_configr   r   r   r    rC   Úis_vqa)
r%   rO   rP   r    rC   rQ   r   rL   r&   r'   s
            €r(   r$   zPix2StructConfig.__init__)  s   ø€ ô 	‰ÑÐrÐ-@ÐUgÑrÐkqÒràÐØˆKÜ�K‰KÐiÔjàÐ ØˆMÜ�K‰KÐmÔnà,>ˆÐ(Ñ)Ø-@ˆÐ)Ñ*Ü/Ñ>°+Ñ>ˆÔÜ3ÑD°mÑDˆÔà&*×&6Ñ&6×&MÑ&MˆÔ#Ø ×,Ñ,×9Ñ9ˆÔØ ×,Ñ,×9Ñ9ˆÔà"4ˆÔØ!2ˆÔà-1×-CÑ-Cˆ×ÑÔ*Ø/3×/EÑ/Eˆ×ÑÔ,àˆ�r)   rO   rP   c                 óP   —  | d|j                  «       |j                  «       dœ|¤ŽS )zÿ
        Instantiate a [`Pix2StructConfig`] (or a derived class) from pix2struct text model configuration and pix2struct
        vision model configuration.

        Returns:
            [`Pix2StructConfig`]: An instance of a configuration object
        )rO   rP   r   )Úto_dict)ÚclsrO   rP   r&   s       r(   Úfrom_text_vision_configsz)Pix2StructConfig.from_text_vision_configsO  s,   € ñ Ðf˜{×2Ñ2Ó4ÀM×DYÑDYÓD[ÑfÐ_eÑfÐfr)   )NNr1   g{®Gáz”?FFT)r5   r6   r7   r8   r9   r$   Úclassmethodr   r?   rU   r<   r=   s   @r(   rI   rI   ÷   sU   ø„ ñ-ð^ €Jð ØØØØØ!Øõ$ðL ðgØ.ðgØ?Uògó ôgr)   rI   )rI   r   r?   N)r8   Úconfiguration_utilsr   Úutilsr   Ú
get_loggerr5   rM   r   r?   rI   Ú__all__r   r)   r(   ú<module>r[      s^   ðñ %å 3Ý ð 
ˆ×	Ñ	˜HÓ	%€ôy
Ð+ô y
ôx`Ð-ô `ôFdgÐ'ô dgòN Q�r)   