Ë
    T^(h«  ã                   ó�   — d Z ddlmZ ddlmZ ddlmZ ddlmZ  ej                  e
«      Z G d„ de«      Z G d	„ d
e«      Zdd
gZy)zLongT5 model configurationé    )ÚMappingé   )ÚPretrainedConfig)ÚOnnxSeq2SeqConfigWithPast)Úloggingc                   ód   ‡ — e Zd ZdZdZdgZdddddœZ	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 	 d
ˆ fd	„	Zˆ xZS )ÚLongT5Configaþ  
    This is the configuration class to store the configuration of a [`LongT5Model`] or a [`FlaxLongT5Model`]. It is
    used to instantiate a LongT5 model according to the specified arguments, defining the model architecture.
    Instantiating a configuration with the defaults will yield a similar configuration to that of the LongT5
    [google/long-t5-local-base](https://huggingface.co/google/long-t5-local-base) architecture.

    Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
    documentation from [`PretrainedConfig`] for more information.

    Arguments:
        vocab_size (`int`, *optional*, defaults to 32128):
            Vocabulary size of the LongT5 model. Defines the number of different tokens that can be represented by the
            `inputs_ids` passed when calling [`LongT5Model`].
        d_model (`int`, *optional*, defaults to 512):
            Size of the encoder layers and the pooler layer.
        d_kv (`int`, *optional*, defaults to 64):
            Size of the key, query, value projections per attention head. `d_kv` has to be equal to `d_model //
            num_heads`.
        d_ff (`int`, *optional*, defaults to 2048):
            Size of the intermediate feed forward layer in each `LongT5Block`.
        num_layers (`int`, *optional*, defaults to 6):
            Number of hidden layers in the Transformer encoder.
        num_decoder_layers (`int`, *optional*):
            Number of hidden layers in the Transformer decoder. Will use the same value as `num_layers` if not set.
        num_heads (`int`, *optional*, defaults to 8):
            Number of attention heads for each attention layer in the Transformer encoder.
        local_radius (`int`, *optional*, defaults to 127)
            Number of tokens to the left/right for each token to locally self-attend in a local attention mechanism.
        global_block_size (`int`, *optional*, defaults to 16)
            Lenght of blocks an input sequence is divided into for a global token representation. Used only for
            `encoder_attention_type = "transient-global"`.
        relative_attention_num_buckets (`int`, *optional*, defaults to 32):
            The number of buckets to use for each attention layer.
        relative_attention_max_distance (`int`, *optional*, defaults to 128):
            The maximum distance of the longer sequences for the bucket separation.
        dropout_rate (`float`, *optional*, defaults to 0.1):
            The ratio for all dropout layers.
        layer_norm_eps (`float`, *optional*, defaults to 1e-6):
            The epsilon used by the layer normalization layers.
        initializer_factor (`float`, *optional*, defaults to 1):
            A factor for initializing all weight matrices (should be kept to 1, used internally for initialization
            testing).
        feed_forward_proj (`string`, *optional*, defaults to `"relu"`):
            Type of feed forward layer to be used. Should be one of `"relu"` or `"gated-gelu"`. LongT5v1.1 uses the
            `"gated-gelu"` feed forward projection. Original LongT5 implementation uses `"gated-gelu"`.
        encoder_attention_type (`string`, *optional*, defaults to `"local"`):
            Type of encoder attention to be used. Should be one of `"local"` or `"transient-global"`, which are
            supported by LongT5 implementation.
        use_cache (`bool`, *optional*, defaults to `True`):
            Whether or not the model should return the last key/values attentions (not used by all models).
    Úlongt5Úpast_key_valuesÚd_modelÚ	num_headsÚ
num_layersÚd_kv)Úhidden_sizeÚnum_attention_headsÚnum_hidden_layersÚhead_dimc                 ó  •— || _         || _        || _        || _        || _        |�|n| j                  | _        || _        || _        |	| _        |
| _	        || _
        || _        || _        || _        || _        || _        || _        | j                  j#                  d«      }|d   | _        |d   dk(  | _        t)        |«      dkD  r|d   dk7  st)        |«      dkD  rt+        d|› d�«      ‚|d	k(  rd
| _        t-        ‰| �\  d|||dœ|¤Ž y )Nú-éÿÿÿÿr   Úgatedé   é   z`feed_forward_proj`: z© is not a valid activation function of the dense layer. Please make sure `feed_forward_proj` is of the format `gated-{ACT_FN}` or `{ACT_FN}`, e.g. 'gated-gelu' or 'relu'z
gated-geluÚgelu_new)Úpad_token_idÚeos_token_idÚis_encoder_decoder© )Ú
vocab_sizer   r   Úd_ffr   Únum_decoder_layersr   Úlocal_radiusÚglobal_block_sizeÚrelative_attention_num_bucketsÚrelative_attention_max_distanceÚdropout_rateÚlayer_norm_epsilonÚinitializer_factorÚfeed_forward_projÚencoder_attention_typeÚ	use_cacheÚsplitÚdense_act_fnÚis_gated_actÚlenÚ
ValueErrorÚsuperÚ__init__)Úselfr   r   r   r    r   r!   r   r"   r#   r$   r%   r&   r'   r(   r)   r   r*   r+   r   r   ÚkwargsÚact_infoÚ	__class__s                          €úm/var/www/skyplay_api_hub/venv/lib/python3.12/site-packages/transformers/models/longt5/configuration_longt5.pyr2   zLongT5Config.__init__Y   sA  ø€ ð0 %ˆŒØˆŒØˆŒ	ØˆŒ	Ø$ˆŒà8JÐ8VÑ"4Ð\`×\kÑ\kˆÔØ"ˆŒØ(ˆÔØ!2ˆÔØ.LˆÔ+Ø/NˆÔ,Ø(ˆÔØ"4ˆÔØ"4ˆÔØ!2ˆÔØ&<ˆÔ#Ø"ˆŒà×)Ñ)×/Ñ/°Ó4ˆØ$ R™LˆÔØ$ Q™K¨7Ñ2ˆÔäˆx‹=˜1Ò ¨!¡°Ò!7¼3¸x»=È1Ò;LÜØ'Ð(9Ð':ð ;)ð )óð ð  Ò,Ø *ˆDÔä‰Ñð 	
Ø%Ø%Ø1ñ	
ð ó		
ó    )i€}  i   é@   i   é   Né   é   é   é    é€   gš™™™™™¹?g�íµ ÷Æ°>g      ð?ÚreluTÚlocalTr   r   )	Ú__name__Ú
__module__Ú__qualname__Ú__doc__Ú
model_typeÚkeys_to_ignore_at_inferenceÚattribute_mapr2   Ú__classcell__)r6   s   @r7   r	   r	      st   ø„ ñ2ðh €JØ#4Ð"5Ðà Ø*Ø)Øñ	€Mð ØØØØØØØØØ')Ø(+ØØØØ ØØ&ØØØ÷+?
ñ ?
r8   r	   c                   óL   — e Zd Zedeeeeef   f   fd„«       Zedefd„«       Zy)ÚLongT5OnnxConfigÚreturnc                 óÂ   — dddœdddœdœ}| j                   rd|d   d<   ddi|d	<   dd
dœ|d<   ndddœ|d	<   dddœ|d<   | j                   r| j                  |d¬«       |S )NÚbatchÚencoder_sequence)r   r   )Ú	input_idsÚattention_maskz past_encoder_sequence + sequencerQ   r   r   Údecoder_input_idsz past_decoder_sequence + sequenceÚdecoder_attention_maskÚdecoder_sequenceÚinputs)Ú	direction)Úuse_pastÚfill_with_past_key_values_)r3   Úcommon_inputss     r7   rU   zLongT5OnnxConfig.inputsœ   s˜   € ð %Ð);Ñ<Ø")Ð.@ÑAñ
ˆð �=Š=Ø1SˆMÐ*Ñ+¨AÑ.Ø23°W°ˆMÐ-Ñ.Ø:AÐFhÑ6iˆMÐ2Ò3à5<ÐASÑ1TˆMÐ-Ñ.Ø:AÐFXÑ6YˆMÐ2Ñ3à�=Š=Ø×+Ñ+¨MÀXÐ+ÔNàÐr8   c                  ó   — y)Né   r   )r3   s    r7   Údefault_onnx_opsetz#LongT5OnnxConfig.default_onnx_opset¯   s   € àr8   N)	rB   rC   rD   Úpropertyr   ÚstrÚintrU   r\   r   r8   r7   rK   rK   ›   sI   „ Øð˜  W¨S°#¨XÑ%6Ð 6Ñ7ò ó ðð$ ð Cò ó ñr8   rK   N)rE   Útypingr   Úconfiguration_utilsr   Úonnxr   Úutilsr   Ú
get_loggerrB   Úloggerr	   rK   Ú__all__r   r8   r7   ú<module>rg      sT   ðñ !å å 3Ý -Ý ð 
ˆ×	Ñ	˜HÓ	%€ô}
Ð#ô }
ô@Ð0ô ð2 Ð-Ð
.�r8   