Ë
    T^(h97  ã                   ó°   — d dl Z d dlmZ d dlmZmZmZmZmZ d dl	Z
ddlmZmZmZ ddlmZ  ej"                  e«      ZdZdd	iZg d
¢Z G d„ de«      ZdgZy)é    N)Úcopyfile)ÚAnyÚDictÚListÚOptionalÚTupleé   )Ú
AddedTokenÚBatchEncodingÚPreTrainedTokenizer)Úloggingu   â–�Ú
vocab_filezsentencepiece.bpe.model)Úar_ARÚcs_CZÚde_DEÚen_XXÚes_XXÚet_EEÚfi_FIÚfr_XXÚgu_INÚhi_INÚit_ITÚja_XXÚkk_KZÚko_KRÚlt_LTÚlv_LVÚmy_MMÚne_NPÚnl_XXÚro_ROÚru_RUÚsi_LKÚtr_TRÚvi_VNÚzh_CNc                   óN  ‡ — e Zd ZU dZeZddgZg Zee	   e
d<   g Zee	   e
d<   	 	 	 	 	 	 	 	 	 	 	 	 d+deeeef      fˆ fd„Zd	„ Zd
„ Zed„ «       Zedefd„«       Zej,                  deddfd„«       Z	 d,dee	   deee	      dedee	   fˆ fd„Z	 d-dee	   deee	      dee	   fd„Z	 d-dee	   deee	      dee	   fd„Zdedee   dee   fd„Zd„ Zdedee   fd„Zd„ Zd„ Zd„ Z d-d ed!ee   de!e   fd"„Z"	 	 	 d.d#ee   ded$eee      dede#f
ˆ fd%„Z$d&„ Z%d'„ Z&d/d(„Z'd)eddfd*„Z(ˆ xZ)S )0ÚMBartTokenizeruT  
    Construct an MBART tokenizer.

    Adapted from [`RobertaTokenizer`] and [`XLNetTokenizer`]. Based on
    [SentencePiece](https://github.com/google/sentencepiece).

    The tokenization method is `<tokens> <eos> <language code>` for source language documents, and `<language code>
    <tokens> <eos>` for target language documents.

    Examples:

    ```python
    >>> from transformers import MBartTokenizer

    >>> tokenizer = MBartTokenizer.from_pretrained("facebook/mbart-large-en-ro", src_lang="en_XX", tgt_lang="ro_RO")
    >>> example_english_phrase = " UN Chief Says There Is No Military Solution in Syria"
    >>> expected_translation_romanian = "Åžeful ONU declarÄƒ cÄƒ nu existÄƒ o soluÅ£ie militarÄƒ Ã®n Siria"
    >>> inputs = tokenizer(example_english_phrase, text_target=expected_translation_romanian, return_tensors="pt")
    ```Ú	input_idsÚattention_maskÚprefix_tokensÚsuffix_tokensNÚsp_model_kwargsc                 ó  •— t        |t        «      rt        |dd¬«      n|}|€i n|| _        t	        j
                  di | j                  ¤Ž| _        | j                  j                  t        |«      «       || _        dddddœ| _	        d| _
        t        | j                  «      | _        t        t        «      D ��ci c]"  \  }}|| j                  |z   | j                  z   “Œ$ c}}| _        | j                  j!                  «       D ��ci c]  \  }}||“Œ
 c}}| _        t        | j                  «      t        | j                  «      z   | j                  z   | j                  d	<   | j                  j%                  | j                  «       | j                  j!                  «       D ��ci c]  \  }}||“Œ
 c}}| _        t)        | j                  j+                  «       «      }|�$|j-                  |D �cg c]	  }||vsŒ|‘Œ c}«       t/        ‰| �`  d|||||||d |
||| j                  d
œ|¤Ž |
�|
nd| _        | j                  | j2                     | _        || _        | j9                  | j2                  «       y c c}}w c c}}w c c}}w c c}w )NTF)ÚlstripÚ
normalizedr   é   é   r	   )ú<s>ú<pad>ú</s>ú<unk>ú<mask>)Ú	bos_tokenÚ	eos_tokenÚ	unk_tokenÚ	sep_tokenÚ	cls_tokenÚ	pad_tokenÚ
mask_tokenÚtokenizer_fileÚsrc_langÚtgt_langÚadditional_special_tokensr.   r   © )Ú
isinstanceÚstrr
   r.   ÚspmÚSentencePieceProcessorÚsp_modelÚLoadr   Úfairseq_tokens_to_idsÚfairseq_offsetÚlenÚsp_model_sizeÚ	enumerateÚFAIRSEQ_LANGUAGE_CODESÚlang_code_to_idÚitemsÚid_to_lang_codeÚupdateÚfairseq_ids_to_tokensÚlistÚkeysÚextendÚsuperÚ__init__Ú	_src_langÚcur_lang_code_idrB   Úset_src_lang_special_tokens)Úselfr   r9   r:   r<   r=   r;   r>   r?   r@   rA   rB   r.   rC   ÚkwargsÚiÚcodeÚkÚvÚ_additional_special_tokensÚtÚ	__class__s                        €új/var/www/skyplay_api_hub/venv/lib/python3.12/site-packages/transformers/models/mbart/tokenization_mbart.pyrZ   zMBartTokenizer.__init__?   sd  ø€ ô& FPÐPZÔ\_ÔE`ŒJ�z¨$¸5ÕAÐfpð 	ð &5Ð%<™rÀ/ˆÔä×2Ñ2ÑJ°T×5IÑ5IÑJˆŒØ�‰×Ñœ3˜z›?Ô+Ø$ˆŒð ./¸ÀAÐPQÑ%RˆÔ"ð  ˆÔä  §¡Ó/ˆÔäNWÔXnÓNo÷ 
ÙCJÀ1ÀdˆD�$×$Ñ$ qÑ(¨4×+>Ñ+>Ñ>Ñ>ó 
ˆÔð 26×1EÑ1E×1KÑ1KÓ1M×N©¨¨A  1¡ÓNˆÔÜ/2°4·=±=Ó/AÄCÈ×H\ÑH\ÓD]Ñ/]Ð`d×`sÑ`sÑ/sˆ×"Ñ" 8Ñ,à×"Ñ"×)Ñ)¨$×*>Ñ*>Ô?Ø7;×7QÑ7Q×7WÑ7WÓ7Y×%Z©t¨q°! a¨¡dÓ%ZˆÔ"Ü%)¨$×*>Ñ*>×*CÑ*CÓ*EÓ%FÐ"à$Ð0à&×-Ñ-Ø5Ö]�q¸ÐB\Ò9\’Ò]ôô 	‰Ñð 	
ØØØØØØØ!ØØØØ&@Ø ×0Ñ0ñ	
ð ò	
ð  &.Ð%9™¸wˆŒØ $× 4Ñ 4°T·^±^Ñ DˆÔØ ˆŒØ×(Ñ(¨¯©Õ8ùóG 
ùó  Oùó &[ùò ^s   Â;'I6ÄI<Æ%JÇ/	JÇ9Jc                 ó~   — | j                   j                  «       }d |d<   | j                  j                  «       |d<   |S )NrI   Úsp_model_proto)Ú__dict__ÚcopyrI   Úserialized_model_proto)r^   Ústates     rg   Ú__getstate__zMBartTokenizer.__getstate__�   s;   € Ø—‘×"Ñ"Ó$ˆØ ˆˆjÑØ"&§-¡-×"FÑ"FÓ"HˆÐÑØˆó    c                 óÊ   — || _         t        | d«      si | _        t        j                  di | j                  ¤Ž| _        | j
                  j                  | j                  «       y )Nr.   rD   )rj   Úhasattrr.   rG   rH   rI   ÚLoadFromSerializedProtori   )r^   Úds     rg   Ú__setstate__zMBartTokenizer.__setstate__“   sQ   € ØˆŒô �tÐ.Ô/Ø#%ˆDÔ ä×2Ñ2ÑJ°T×5IÑ5IÑJˆŒØ�‰×-Ñ-¨d×.AÑ.AÕBro   c                 óx   — t        | j                  «      t        | j                  «      z   | j                  z   dz   S )Nr2   )rM   rI   rQ   rL   ©r^   s    rg   Ú
vocab_sizezMBartTokenizer.vocab_size�   s2   € ä�4—=‘=Ó!¤C¨×(<Ñ(<Ó$=Ñ=À×@SÑ@SÑSÐVWÑWÐWro   Úreturnc                 ó   — | j                   S ©N)r[   rv   s    rg   rA   zMBartTokenizer.src_lang¡   s   € à�~‰~Ðro   Únew_src_langc                 óH   — || _         | j                  | j                   «       y rz   )r[   r]   )r^   r{   s     rg   rA   zMBartTokenizer.src_lang¥   s   € à%ˆŒØ×(Ñ(¨¯©Õ8ro   Útoken_ids_0Útoken_ids_1Úalready_has_special_tokensc                 ó  •— |rt         ‰| �  ||d¬«      S dgt        | j                  «      z  }dgt        | j                  «      z  }|€|dgt        |«      z  z   |z   S |dgt        |«      z  z   dgt        |«      z  z   |z   S )aÄ  
        Retrieve sequence ids from a token list that has no special tokens added. This method is called when adding
        special tokens using the tokenizer `prepare_for_model` method.

        Args:
            token_ids_0 (`List[int]`):
                List of IDs.
            token_ids_1 (`List[int]`, *optional*):
                Optional second list of IDs for sequence pairs.
            already_has_special_tokens (`bool`, *optional*, defaults to `False`):
                Whether or not the token list is already formatted with special tokens for the model.

        Returns:
            `List[int]`: A list of integers in the range [0, 1]: 1 for a special token, 0 for a sequence token.
        T)r}   r~   r   r2   r   )rY   Úget_special_tokens_maskrM   r,   r-   )r^   r}   r~   r   Úprefix_onesÚsuffix_onesrf   s         €rg   r�   z&MBartTokenizer.get_special_tokens_maskª   s§   ø€ ñ& &Ü‘7Ñ2Ø'°[Ð]að 3ó ð ð �cœC × 2Ñ 2Ó3Ñ3ˆØ�cœC × 2Ñ 2Ó3Ñ3ˆØÐØ 1 #¬¨KÓ(8Ñ"8Ñ9¸KÑGÐGØ˜q˜c¤C¨Ó$4Ñ4Ñ5¸!¸¼sÀ;Ó?OÑ9OÑPÐS^Ñ^Ð^ro   c                 ó|   — |€| j                   |z   | j                  z   S | j                   |z   |z   | j                  z   S )ab  
        Build model inputs from a sequence or a pair of sequence for sequence classification tasks by concatenating and
        adding special tokens. An MBART sequence has the following format, where `X` represents the sequence:

        - `input_ids` (for encoder) `X [eos, src_lang_code]`
        - `decoder_input_ids`: (for decoder) `X [eos, tgt_lang_code]`

        BOS is never used. Pairs of sequences are not the expected use case, but they will be handled without a
        separator.

        Args:
            token_ids_0 (`List[int]`):
                List of IDs to which the special tokens will be added.
            token_ids_1 (`List[int]`, *optional*):
                Optional second list of IDs for sequence pairs.

        Returns:
            `List[int]`: List of [input IDs](../glossary#input-ids) with the appropriate special tokens.
        )r,   r-   )r^   r}   r~   s      rg   Ú build_inputs_with_special_tokensz/MBartTokenizer.build_inputs_with_special_tokensÈ   sG   € ð, ÐØ×%Ñ%¨Ñ3°d×6HÑ6HÑHÐHà×!Ñ! KÑ/°+Ñ=À×@RÑ@RÑRÐRro   c                 ó    — | j                   g}| j                  g}|€t        ||z   |z   «      dgz  S t        ||z   |z   |z   |z   |z   «      dgz  S )aË  
        Create a mask from the two sequences passed to be used in a sequence-pair classification task. mBART does not
        make use of token type ids, therefore a list of zeros is returned.

        Args:
            token_ids_0 (`List[int]`):
                List of IDs.
            token_ids_1 (`List[int]`, *optional*):
                Optional second list of IDs for sequence pairs.

        Returns:
            `List[int]`: List of zeros.

        r   )Úsep_token_idÚcls_token_idrM   )r^   r}   r~   ÚsepÚclss        rg   Ú$create_token_type_ids_from_sequencesz3MBartTokenizer.create_token_type_ids_from_sequencesã   sm   € ð$ × Ñ Ð!ˆØ× Ñ Ð!ˆàÐÜ�s˜[Ñ(¨3Ñ.Ó/°1°#Ñ5Ð5Ü�3˜Ñ$ sÑ*¨SÑ0°;Ñ>ÀÑDÓEÈÈÑKÐKro   Úreturn_tensorsrA   rB   c                 óv   — |�|€t        d«      ‚|| _         | |fd|dœ|¤Ž}| j                  |«      }||d<   |S )zIUsed by translation pipeline, to prepare inputs for the generate functionzATranslation requires a `src_lang` and a `tgt_lang` for this modelT)Úadd_special_tokensrŒ   Úforced_bos_token_id)Ú
ValueErrorrA   Úconvert_tokens_to_ids)r^   Ú
raw_inputsrŒ   rA   rB   Úextra_kwargsÚinputsÚtgt_lang_ids           rg   Ú_build_translation_inputsz(MBartTokenizer._build_translation_inputsü   sY   € ð Ð˜xÐ/ÜÐ`ÓaÐaØ ˆŒÙ�jÐi°TÈ.ÑiÐ\hÑiˆØ×0Ñ0°Ó:ˆØ(3ˆÐ$Ñ%Øˆro   c                 óª   — t        | j                  «      D �ci c]  }| j                  |«      |“Œ }}|j                  | j                  «       |S c c}w rz   )Úrangerw   Úconvert_ids_to_tokensrT   Úadded_tokens_encoder)r^   r`   Úvocabs      rg   Ú	get_vocabzMBartTokenizer.get_vocab  sK   € Ü;@ÀÇÁÓ;QÖR°a�×+Ñ+¨AÓ.°Ñ1ÐRˆÐRØ�‰�T×.Ñ.Ô/Øˆùò Ss   ˜AÚtextc                 óD   — | j                   j                  |t        ¬«      S )N)Úout_type)rI   ÚencoderF   )r^   r�   s     rg   Ú	_tokenizezMBartTokenizer._tokenize  s   € Ø�}‰}×#Ñ# D´3Ð#Ó7Ð7ro   c                 ó¬   — || j                   v r| j                   |   S | j                  j                  |«      }|r|| j                  z   S | j                  S )z0Converts a token (str) in an id using the vocab.)rK   rI   Ú	PieceToIdrL   Úunk_token_id)r^   ÚtokenÚspm_ids      rg   Ú_convert_token_to_idz#MBartTokenizer._convert_token_to_id  sU   € à�D×.Ñ.Ñ.Ø×-Ñ-¨eÑ4Ð4Ø—‘×(Ñ(¨Ó/ˆñ 06ˆv˜×+Ñ+Ñ+ÐL¸4×;LÑ;LÐLro   c                 óŒ   — || j                   v r| j                   |   S | j                  j                  || j                  z
  «      S )z=Converts an index (integer) in a token (str) using the vocab.)rU   rI   Ú	IdToPiecerL   )r^   Úindexs     rg   Ú_convert_id_to_tokenz#MBartTokenizer._convert_id_to_token  sA   € à�D×.Ñ.Ñ.Ø×-Ñ-¨eÑ4Ð4Ø�}‰}×&Ñ& u¨t×/BÑ/BÑ'BÓCÐCro   c                 ól   — dj                  |«      j                  t        d«      j                  «       }|S )zIConverts a sequence of tokens (strings for sub-words) in a single string.Ú ú )ÚjoinÚreplaceÚSPIECE_UNDERLINEÚstrip)r^   ÚtokensÚ
out_strings      rg   Úconvert_tokens_to_stringz'MBartTokenizer.convert_tokens_to_string  s,   € à—W‘W˜V“_×,Ñ,Ô-=¸sÓC×IÑIÓKˆ
ØÐro   Úsave_directoryÚfilename_prefixc                 óæ  — t         j                  j                  |«      st        j	                  d|› d�«       y t         j                  j                  ||r|dz   ndt        d   z   «      }t         j                  j                  | j                  «      t         j                  j                  |«      k7  rBt         j                  j                  | j                  «      rt        | j                  |«       |fS t         j                  j                  | j                  «      sCt        |d«      5 }| j                  j                  «       }|j                  |«       d d d «       |fS |fS # 1 sw Y   |fS xY w)NzVocabulary path (z) should be a directoryú-r­   r   Úwb)ÚosÚpathÚisdirÚloggerÚerrorr¯   ÚVOCAB_FILES_NAMESÚabspathr   Úisfiler   ÚopenrI   rl   Úwrite)r^   r¶   r·   Úout_vocab_fileÚfiÚcontent_spiece_models         rg   Úsave_vocabularyzMBartTokenizer.save_vocabulary$  s%  € Ü�w‰w�}‰}˜^Ô,Ü�L‰LÐ,¨^Ð,<Ð<SÐTÔUØÜŸ™Ÿ™Ø±o˜_¨sÒ2È2ÔQbÐcoÑQpÑpó
ˆô �7‰7�?‰?˜4Ÿ?™?Ó+¬r¯w©w¯©¸~Ó/NÒNÔSU×SZÑSZ×SaÑSaÐbf×bqÑbqÔSrÜ�T—_‘_ nÔ5ð Ð Ð ô —‘—‘ §¡Ô0Ü�n dÓ+ð /¨rØ'+§}¡}×'KÑ'KÓ'MÐ$Ø—‘Ð-Ô.÷/ð Ð Ð �Ð Ð ÷	/ð Ð Ð ús   Ä+,E%Å%E0Ú	src_textsÚ	tgt_textsc                 óB   •— || _         || _        t        ‰| �  ||fi |¤ŽS rz   )rA   rB   rY   Úprepare_seq2seq_batch)r^   rÉ   rA   rÊ   rB   r_   rf   s         €rg   rÌ   z$MBartTokenizer.prepare_seq2seq_batch5  s*   ø€ ð !ˆŒØ ˆŒÜ‰wÑ,¨Y¸	ÑLÀVÑLÐLro   c                 ó8   — | j                  | j                  «      S rz   )r]   rA   rv   s    rg   Ú_switch_to_input_modez$MBartTokenizer._switch_to_input_modeA  ó   € Ø×/Ñ/°·±Ó>Ð>ro   c                 ó8   — | j                  | j                  «      S rz   )Úset_tgt_lang_special_tokensrB   rv   s    rg   Ú_switch_to_target_modez%MBartTokenizer._switch_to_target_modeD  rÏ   ro   c                 ót   — | j                   |   | _        g | _        | j                  | j                  g| _        y)z_Reset the special tokens to the source lang setting. No prefix and suffix=[eos, src_lang_code].N©rQ   Úcur_lang_coder,   Úeos_token_idr-   )r^   rA   s     rg   r]   z*MBartTokenizer.set_src_lang_special_tokensG  s6   € à!×1Ñ1°(Ñ;ˆÔØˆÔØ"×/Ñ/°×1CÑ1CÐDˆÕro   Úlangc                 ót   — | j                   |   | _        g | _        | j                  | j                  g| _        y)zcReset the special tokens to the target language setting. No prefix and suffix=[eos, tgt_lang_code].NrÔ   )r^   r×   s     rg   rÑ   z*MBartTokenizer.set_tgt_lang_special_tokensM  s6   € à!×1Ñ1°$Ñ7ˆÔØˆÔØ"×/Ñ/°×1CÑ1CÐDˆÕro   )r4   r6   r6   r4   r7   r5   r8   NNNNN)NFrz   )r   Nr"   )rx   N)*Ú__name__Ú
__module__Ú__qualname__Ú__doc__rÀ   Úvocab_files_namesÚmodel_input_namesr,   r   ÚintÚ__annotations__r-   r   r   rF   r   rZ   rn   rt   Úpropertyrw   rA   ÚsetterÚboolr�   r…   r‹   r–   rœ   r¡   r§   r«   rµ   r   rÈ   r   rÌ   rÎ   rÒ   r]   rÑ   Ú__classcell__)rf   s   @rg   r)   r)   $   s…  ø… ñð( *ÐØ$Ð&6Ð7Ðà!€M�4˜‘9Ó!Ø!€M�4˜‘9Ó!ð
 ØØØØØØØØØØ48Ø"&ñL9ð " $ s¨C x¡.Ñ1õL9ò\òCð ñXó ðXð ð˜#ò ó ðð ‡_�_ð9 Sð 9¨Tò 9ó ð9ð
 sxñ_Ø ™9ð_Ø3;¸DÀ¹IÑ3Fð_Økoð_à	ˆc‰õ_ð> JNñSØ ™9ðSØ3;¸DÀ¹IÑ3FðSà	ˆc‰óSð8 JNñLØ ™9ðLØ3;¸DÀ¹IÑ3FðLà	ˆc‰óLð2
Ø*-ð
Ø9AÀ#¹ð
ØRZÐ[^ÑR_ó
òð
8˜cð 8 d¨3¡ió 8òMòDòñ
!¨cð !ÀHÈSÁMð !Ð]bÐcfÑ]gó !ð(  Ø)-Øñ
Mà˜‘9ð
Mð ð
Mð ˜D ™IÑ&ð	
Mð
 ð
Mð 
õ
Mò?ò?óEðE°ð E¸÷ Ero   r)   )r»   Úshutilr   Útypingr   r   r   r   r   ÚsentencepiecerG   Útokenization_utilsr
   r   r   Úutilsr   Ú
get_loggerrÙ   r¾   r±   rÀ   rP   r)   Ú__all__rD   ro   rg   ú<module>rì      sj   ðó  
Ý ß 3Õ 3ã ç PÑ PÝ ð 
ˆ×	Ñ	˜HÓ	%€àÐ à!Ð#<Ð=Ð ò {Ð ômEÐ(ô mEð`	 Ð
�ro   