Ë
    T^(hû  ã                   ó`   — d Z ddlmZ ddlmZ  ej
                  e«      Z G d„ de«      ZdgZ	y)zUDOP model configurationé   )ÚPretrainedConfig)Úloggingc                   óx   ‡ — e Zd ZdZdZdgZddddœZdd	d
ddddddddiddiddigddddddddd	dddfˆ fd„	Zˆ xZS )Ú
UdopConfigaÕ  
    This is the configuration class to store the configuration of a [`UdopForConditionalGeneration`]. It is used to
    instantiate a UDOP model according to the specified arguments, defining the model architecture. Instantiating a
    configuration with the defaults will yield a similar configuration to that of the UDOP
    [microsoft/udop-large](https://huggingface.co/microsoft/udop-large) architecture.

    Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
    documentation from [`PretrainedConfig`] for more information.

    Arguments:
        vocab_size (`int`, *optional*, defaults to 33201):
            Vocabulary size of the UDOP model. Defines the number of different tokens that can be represented by the
            `inputs_ids` passed when calling [`UdopForConditionalGeneration`].
        d_model (`int`, *optional*, defaults to 1024):
            Size of the encoder layers and the pooler layer.
        d_kv (`int`, *optional*, defaults to 64):
            Size of the key, query, value projections per attention head. The `inner_dim` of the projection layer will
            be defined as `num_heads * d_kv`.
        d_ff (`int`, *optional*, defaults to 4096):
            Size of the intermediate feed forward layer in each `UdopBlock`.
        num_layers (`int`, *optional*, defaults to 24):
            Number of hidden layers in the Transformer encoder and decoder.
        num_decoder_layers (`int`, *optional*):
            Number of hidden layers in the Transformer decoder. Will use the same value as `num_layers` if not set.
        num_heads (`int`, *optional*, defaults to 16):
            Number of attention heads for each attention layer in the Transformer encoder and decoder.
        relative_attention_num_buckets (`int`, *optional*, defaults to 32):
            The number of buckets to use for each attention layer.
        relative_attention_max_distance (`int`, *optional*, defaults to 128):
            The maximum distance of the longer sequences for the bucket separation.
        relative_bias_args (`List[dict]`, *optional*, defaults to `[{'type': '1d'}, {'type': 'horizontal'}, {'type': 'vertical'}]`):
            A list of dictionaries containing the arguments for the relative bias layers.
        dropout_rate (`float`, *optional*, defaults to 0.1):
            The ratio for all dropout layers.
        layer_norm_epsilon (`float`, *optional*, defaults to 1e-06):
            The epsilon used by the layer normalization layers.
        initializer_factor (`float`, *optional*, defaults to 1.0):
            A factor for initializing all weight matrices (should be kept to 1, used internally for initialization
            testing).
        feed_forward_proj (`string`, *optional*, defaults to `"relu"`):
            Type of feed forward layer to be used. Should be one of `"relu"` or `"gated-gelu"`. Udopv1.1 uses the
            `"gated-gelu"` feed forward projection. Original Udop uses `"relu"`.
        is_encoder_decoder (`bool`, *optional*, defaults to `True`):
            Whether the model should behave as an encoder/decoder or not.
        use_cache (`bool`, *optional*, defaults to `True`):
            Whether or not the model should return the last key/values attentions (not used by all models).
        pad_token_id (`int`, *optional*, defaults to 0):
            The id of the padding token in the vocabulary.
        eos_token_id (`int`, *optional*, defaults to 1):
            The id of the end-of-sequence token in the vocabulary.
        max_2d_position_embeddings (`int`, *optional*, defaults to 1024):
            The maximum absolute position embeddings for relative position encoding.
        image_size (`int`, *optional*, defaults to 224):
            The size of the input images.
        patch_size (`int`, *optional*, defaults to 16):
            The patch size used by the vision encoder.
        num_channels (`int`, *optional*, defaults to 3):
            The number of channels in the input images.
    ÚudopÚpast_key_valuesÚd_modelÚ	num_headsÚ
num_layers)Úhidden_sizeÚnum_attention_headsÚnum_hidden_layersi±�  i   é@   i   é   Né   é    é€   ÚtypeÚ1dÚ
horizontalÚverticalgš™™™™™¹?g�íµ ÷Æ°>g      ð?ÚreluTé    é   éà   r   c                 óR  •— || _         || _        || _        || _        || _        |�|n| j                  | _        || _        || _        |	| _        || _	        || _
        || _        || _        || _        || _        || _        || _        || _        t%        |
t&        «      st)        d«      ‚|
| _        | j                  j-                  d«      }|d   | _        |d   dk(  | _        t3        |«      dkD  r|d   dk7  st3        |«      dkD  rt5        d|› d	�«      ‚t7        ‰| �p  d|||d
œ|¤Ž y )Nz6`relative_bias_args` should be a list of dictionaries.ú-éÿÿÿÿr   Úgatedr   é   z`feed_forward_proj`: z¨ is not a valid activation function of the dense layer.Please make sure `feed_forward_proj` is of the format `gated-{ACT_FN}` or `{ACT_FN}`, e.g. 'gated-gelu' or 'relu')Úpad_token_idÚeos_token_idÚis_encoder_decoder© )Ú
vocab_sizer	   Úd_kvÚd_ffr   Únum_decoder_layersr
   Úrelative_attention_num_bucketsÚrelative_attention_max_distanceÚdropout_rateÚlayer_norm_epsilonÚinitializer_factorÚfeed_forward_projÚ	use_cacheÚmax_2d_position_embeddingsÚ
image_sizeÚ
patch_sizeÚnum_channelsÚ
isinstanceÚlistÚ	TypeErrorÚrelative_bias_argsÚsplitÚdense_act_fnÚis_gated_actÚlenÚ
ValueErrorÚsuperÚ__init__)Úselfr%   r	   r&   r'   r   r(   r
   r)   r*   r7   r+   r,   r-   r.   r#   r/   r!   r"   r0   r1   r2   r3   ÚkwargsÚact_infoÚ	__class__s                            €úi/var/www/skyplay_api_hub/venv/lib/python3.12/site-packages/transformers/models/udop/configuration_udop.pyr>   zUdopConfig.__init__Y   s[  ø€ ð4 %ˆŒØˆŒØˆŒ	ØˆŒ	Ø$ˆŒà"4Ð"@ÑÀdÇoÁoð 	Ôð #ˆŒØ.LˆÔ+Ø/NˆÔ,Ø(ˆÔØ"4ˆÔØ"4ˆÔØ!2ˆÔØ"ˆŒð +EˆÔ'Ø$ˆŒØ$ˆŒØ(ˆÔÜÐ,¬dÔ3ÜÐTÓUÐUØ"4ˆÔà×)Ñ)×/Ñ/°Ó4ˆØ$ R™LˆÔØ$ Q™K¨7Ñ2ˆÔäˆx‹=˜1Ò ¨!¡°Ò!7¼3¸x»=È1Ò;LÜØ'Ð(9Ð':ð ;)ð )óð ô 	‰Ñð 	
Ø%Ø%Ø1ñ	
ð ó		
ó    )	Ú__name__Ú
__module__Ú__qualname__Ú__doc__Ú
model_typeÚkeys_to_ignore_at_inferenceÚattribute_mapr>   Ú__classcell__)rB   s   @rC   r   r      s‹   ø„ ñ:ðx €JØ#4Ð"5ÐØ$-ÀkÐhtÑu€Mð ØØØØØØØ')Ø(+Ø# T˜N¨V°\Ð,BÀVÈZÐDXÐYØØØØ ØØØØØ#'ØØØ÷/D
ñ D
rD   r   N)
rH   Úconfiguration_utilsr   Úutilsr   Ú
get_loggerrE   Úloggerr   Ú__all__r$   rD   rC   ú<module>rR      s=   ðñ å 3Ý ð 
ˆ×	Ñ	˜HÓ	%€ôE
Ð!ô E
ðP ˆ.�rD   