Ë
    T^(hB  ã                   ó�   — d Z ddlmZ ddlmZ ddlmZ ddlmZ  ej                  e
«      Z G d„ de«      Z G d	„ d
e«      Zdd
gZy)zmT5 model configurationé    )ÚMappingé   )ÚPretrainedConfig)ÚOnnxSeq2SeqConfigWithPast)Úloggingc                   óf   ‡ — e Zd ZdZdZdgZdddddœZ	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 d
ˆ fd	„	Zˆ xZS )Ú	MT5Configa7  
    This is the configuration class to store the configuration of a [`MT5Model`] or a [`TFMT5Model`]. It is used to
    instantiate a mT5 model according to the specified arguments, defining the model architecture. Instantiating a
    configuration with the defaults will yield a similar configuration to that of the mT5
    [google/mt5-small](https://huggingface.co/google/mt5-small) architecture.

    Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
    documentation from [`PretrainedConfig`] for more information.

    Arguments:
        vocab_size (`int`, *optional*, defaults to 250112):
            Vocabulary size of the T5 model. Defines the number of different tokens that can be represented by the
            `inputs_ids` passed when calling [`T5Model`] or [`TFT5Model`].
        d_model (`int`, *optional*, defaults to 512):
            Size of the encoder layers and the pooler layer.
        d_kv (`int`, *optional*, defaults to 64):
            Size of the key, query, value projections per attention head. In the conventional context, it is typically expected that `d_kv` has to be equal to `d_model // num_heads`.
            But in the architecture of mt5-small, `d_kv` is not equal to `d_model //num_heads`. The `inner_dim` of the projection layer will be defined as `num_heads * d_kv`.
        d_ff (`int`, *optional*, defaults to 1024):
            Size of the intermediate feed forward layer in each `T5Block`.
        num_layers (`int`, *optional*, defaults to 8):
            Number of hidden layers in the Transformer encoder.
        num_decoder_layers (`int`, *optional*):
            Number of hidden layers in the Transformer decoder. Will use the same value as `num_layers` if not set.
        num_heads (`int`, *optional*, defaults to 6):
            Number of attention heads for each attention layer in the Transformer encoder.
        relative_attention_num_buckets (`int`, *optional*, defaults to 32):
            The number of buckets to use for each attention layer.
        relative_attention_max_distance (`int`, *optional*, defaults to 128):
            The maximum distance of the longer sequences for the bucket separation.
        dropout_rate (`float`, *optional*, defaults to 0.1):
            The ratio for all dropout layers.
        classifier_dropout (`float`, *optional*, defaults to 0.0):
            The dropout ratio for classifier.
        layer_norm_eps (`float`, *optional*, defaults to 1e-6):
            The epsilon used by the layer normalization layers.
        initializer_factor (`float`, *optional*, defaults to 1):
            A factor for initializing all weight matrices (should be kept to 1, used internally for initialization
            testing).
        feed_forward_proj (`string`, *optional*, defaults to `"gated-gelu"`):
            Type of feed forward layer to be used. Should be one of `"relu"` or `"gated-gelu"`.
        use_cache (`bool`, *optional*, defaults to `True`):
            Whether or not the model should return the last key/values attentions (not used by all models).
    Úmt5Úpast_key_valuesÚd_modelÚ	num_headsÚ
num_layersÚd_kv)Úhidden_sizeÚnum_attention_headsÚnum_hidden_layersÚhead_dimc           
      ó  •— || _         || _        || _        || _        || _        |�|n| j                  | _        || _        || _        |	| _        |
| _	        || _
        || _        || _        || _        || _        | j                  j                  d«      }|d   | _        |d   dk(  | _        t%        |«      dkD  r|d   dk7  st%        |«      dkD  rt'        d|› d�«      ‚|d	k(  rd
| _        t)        ‰| �T  d||||||dœ|¤Ž y )Nú-éÿÿÿÿr   Úgatedé   é   z`feed_forward_proj`: z© is not a valid activation function of the dense layer. Please make sure `feed_forward_proj` is of the format `gated-{ACT_FN}` or `{ACT_FN}`, e.g. 'gated-gelu' or 'relu'ú
gated-geluÚgelu_new)Úis_encoder_decoderÚtokenizer_classÚtie_word_embeddingsÚpad_token_idÚeos_token_idÚdecoder_start_token_id© )Ú
vocab_sizer   r   Úd_ffr   Únum_decoder_layersr   Úrelative_attention_num_bucketsÚrelative_attention_max_distanceÚdropout_rateÚclassifier_dropoutÚlayer_norm_epsilonÚinitializer_factorÚfeed_forward_projÚ	use_cacheÚsplitÚdense_act_fnÚis_gated_actÚlenÚ
ValueErrorÚsuperÚ__init__)Úselfr#   r   r   r$   r   r%   r   r&   r'   r(   r*   r+   r,   r   r-   r   r   r   r    r!   r)   ÚkwargsÚact_infoÚ	__class__s                           €úg/var/www/skyplay_api_hub/venv/lib/python3.12/site-packages/transformers/models/mt5/configuration_mt5.pyr4   zMT5Config.__init__R   s<  ø€ ð2 %ˆŒØˆŒØˆŒ	ØˆŒ	Ø$ˆŒà"4Ð"@ÑÀdÇoÁoð 	Ôð #ˆŒØ.LˆÔ+Ø/NˆÔ,Ø(ˆÔØ"4ˆÔØ"4ˆÔØ"4ˆÔØ!2ˆÔØ"ˆŒà×)Ñ)×/Ñ/°Ó4ˆØ$ R™LˆÔØ$ Q™K¨7Ñ2ˆÔäˆx‹=˜1Ò ¨!¡°Ò!7¼3¸x»=È1Ò;LÜØ'Ð(9Ð':ð ;)ð )óð ð  Ò,Ø *ˆDÔä‰Ñð 	
Ø1Ø+Ø 3Ø%Ø%Ø#9ñ	
ð ó	
ó    )i Ñ i   é@   i   é   Né   é    é€   gš™™™™™¹?g�íµ ÷Æ°>g      ð?r   TTÚT5TokenizerFr   r   r   g        )	Ú__name__Ú
__module__Ú__qualname__Ú__doc__Ú
model_typeÚkeys_to_ignore_at_inferenceÚattribute_mapr4   Ú__classcell__)r8   s   @r9   r	   r	      sy   ø„ ñ+ðZ €JØ#4Ð"5Ðà Ø*Ø)Øñ	€Mð ØØØØØØØ')Ø(+ØØØØ&ØØØ%Ø!ØØØ Ø÷-B
ñ B
r:   r	   c                   ób   — e Zd Zedeeeeef   f   fd„«       Zedefd„«       Zede	fd„«       Z
y)ÚMT5OnnxConfigÚreturnc                 óÂ   — dddœdddœdœ}| j                   rd|d   d<   ddi|d	<   dd
dœ|d<   ndddœ|d	<   dddœ|d<   | j                   r| j                  |d¬«       |S )NÚbatchÚencoder_sequence)r   r   )Ú	input_idsÚattention_maskz past_encoder_sequence + sequencerP   r   r   Údecoder_input_idsz past_decoder_sequence + sequenceÚdecoder_attention_maskÚdecoder_sequenceÚinputs)Ú	direction)Úuse_pastÚfill_with_past_key_values_)r5   Úcommon_inputss     r9   rT   zMT5OnnxConfig.inputs˜   s˜   € ð %Ð);Ñ<Ø")Ð.@ÑAñ
ˆð �=Š=Ø1SˆMÐ*Ñ+¨AÑ.Ø23°W°ˆMÐ-Ñ.Ø:AÐFhÑ6iˆMÐ2Ò3à5<ÐASÑ1TˆMÐ-Ñ.Ø:AÐFXÑ6YˆMÐ2Ñ3à�=Š=Ø×+Ñ+¨MÀXÐ+ÔNàÐr:   c                  ó   — y)Né   r"   ©r5   s    r9   Údefault_onnx_opsetz MT5OnnxConfig.default_onnx_opset¬   s   € ð r:   c                  ó   — y)Ngü©ñÒMb@?r"   r[   s    r9   Úatol_for_validationz!MT5OnnxConfig.atol_for_validation±   s   € àr:   N)rA   rB   rC   Úpropertyr   ÚstrÚintrT   r\   Úfloatr^   r"   r:   r9   rJ   rJ   —   sd   „ Øð˜  W¨S°#¨XÑ%6Ð 6Ñ7ò ó ðð$ ð Cò ó ðð ð Uò ó ñr:   rJ   N)rD   Útypingr   Úconfiguration_utilsr   Úonnxr   Úutilsr   Ú
get_loggerrA   Úloggerr	   rJ   Ú__all__r"   r:   r9   ú<module>rj      sS   ðñ å å 3Ý -Ý ð 
ˆ×	Ñ	˜HÓ	%€ôy
Ð ô y
ôxÐ-ô ð> ˜Ð
(�r:   