Ë
    S^(h   ã                   óÆ   — d Z ddlZddlmZ ddlmZmZmZ ddlm	Z	 ddl
mZ ddlmZmZ  e«       rd	d
lmZ ndZ ej"                  e«      ZdddœZdZ G d„ de«      ZdgZy)z$Tokenization classes for FNet model.é    N)Úcopyfile)ÚListÚOptionalÚTupleé   )Ú
AddedToken)ÚPreTrainedTokenizerFast)Úis_sentencepiece_availableÚloggingé   )ÚFNetTokenizerzspiece.modelztokenizer.json)Ú
vocab_fileÚtokenizer_fileu   â–�c                   óà   ‡ — e Zd ZdZeZddgZeZ	 	 	 	 	 	 	 	 	 	 dˆ fd„	Z	e
defd„«       Z	 ddee   deee      dee   fd	„Z	 ddee   deee      dee   fd
„Zddedee   dee   fd„Zˆ xZS )ÚFNetTokenizerFastaR	  
    Construct a "fast" FNetTokenizer (backed by HuggingFace's *tokenizers* library). Adapted from
    [`AlbertTokenizerFast`]. Based on
    [Unigram](https://huggingface.co/docs/tokenizers/python/latest/components.html?highlight=unigram#models). This
    tokenizer inherits from [`PreTrainedTokenizerFast`] which contains most of the main methods. Users should refer to
    this superclass for more information regarding those methods

    Args:
        vocab_file (`str`):
            [SentencePiece](https://github.com/google/sentencepiece) file (generally has a *.spm* extension) that
            contains the vocabulary necessary to instantiate a tokenizer.
        do_lower_case (`bool`, *optional*, defaults to `False`):
            Whether or not to lowercase the input when tokenizing.
        remove_space (`bool`, *optional*, defaults to `True`):
            Whether or not to strip the text when tokenizing (removing excess spaces before and after the string).
        keep_accents (`bool`, *optional*, defaults to `True`):
            Whether or not to keep accents when tokenizing.
        unk_token (`str`, *optional*, defaults to `"<unk>"`):
            The unknown token. A token that is not in the vocabulary cannot be converted to an ID and is set to be this
            token instead.
        sep_token (`str`, *optional*, defaults to `"[SEP]"`):
            The separator token, which is used when building a sequence from multiple sequences, e.g. two sequences for
            sequence classification or for a text and a question for question answering. It is also used as the last
            token of a sequence built with special tokens.
        pad_token (`str`, *optional*, defaults to `"<pad>"`):
            The token used for padding, for example when batching sequences of different lengths.
        cls_token (`str`, *optional*, defaults to `"[CLS]"`):
            The classifier token which is used when doing sequence classification (classification of the whole sequence
            instead of per-token classification). It is the first token of the sequence when built with special tokens.
        mask_token (`str`, *optional*, defaults to `"[MASK]"`):
            The token used for masking values. This is the token used when training this model with masked language
            modeling. This is the token which the model will try to predict.
    Ú	input_idsÚtoken_type_idsc                 ó2  •— t        |
t        «      rt        |
dd¬«      n|
}
t        |	t        «      rt        |	dd¬«      n|	}	t        |t        «      rt        |dd¬«      n|}t        ‰| �  |f||||||||	|
dœ	|¤Ž || _        || _        || _        || _        y )NTF)ÚlstripÚrstrip)	r   Údo_lower_caseÚremove_spaceÚkeep_accentsÚ	unk_tokenÚ	sep_tokenÚ	pad_tokenÚ	cls_tokenÚ
mask_token)	Ú
isinstanceÚstrr   ÚsuperÚ__init__r   r   r   r   )Úselfr   r   r   r   r   r   r   r   r   r   ÚkwargsÚ	__class__s               €úm/var/www/skyplay_api_hub/venv/lib/python3.12/site-packages/transformers/models/fnet/tokenization_fnet_fast.pyr"   zFNetTokenizerFast.__init__M   s¶   ø€ ô  KUÐU_ÔadÔJe”Z 
°4ÀÕFÐkuˆ
ÜISÐT]Ô_bÔIc”J˜y°¸uÕEÐirˆ	ÜISÐT]Ô_bÔIc”J˜y°¸uÕEÐirˆ	ä‰ÑØð	
à)Ø'Ø%Ø%ØØØØØ!ñ	
ð ò	
ð +ˆÔØ(ˆÔØ(ˆÔØ$ˆ�ó    Úreturnc                 óp   — | j                   r)t        j                  j                  | j                   «      S dS )NF)r   ÚosÚpathÚisfile)r#   s    r&   Úcan_save_slow_tokenizerz)FNetTokenizerFast.can_save_slow_tokenizert   s$   € à26·/²/Œr�w‰w�~‰~˜dŸo™oÓ.ÐLÀuÐLr'   Útoken_ids_0Útoken_ids_1c                 óf   — | j                   g}| j                  g}|€||z   |z   S ||z   |z   |z   |z   S )a–  
        Build model inputs from a sequence or a pair of sequence for sequence classification tasks by concatenating and
        adding special tokens. An FNet sequence has the following format:

        - single sequence: `[CLS] X [SEP]`
        - pair of sequences: `[CLS] A [SEP] B [SEP]`

        Args:
            token_ids_0 (`List[int]`):
                List of IDs to which the special tokens will be added
            token_ids_1 (`List[int]`, *optional*):
                Optional second list of IDs for sequence pairs.

        Returns:
            `List[int]`: list of [input IDs](../glossary#input-ids) with the appropriate special tokens.
        )Úsep_token_idÚcls_token_id©r#   r.   r/   ÚsepÚclss        r&   Ú build_inputs_with_special_tokensz2FNetTokenizerFast.build_inputs_with_special_tokensx   sP   € ð& × Ñ Ð!ˆØ× Ñ Ð!ˆØÐØ˜Ñ$ sÑ*Ð*Ø�[Ñ  3Ñ&¨Ñ4°sÑ:Ð:r'   c                 ó´   — | j                   g}| j                  g}|€t        ||z   |z   «      dgz  S t        ||z   |z   «      dgz  t        ||z   «      dgz  z   S )aÃ  
        Creates a mask from the two sequences passed to be used in a sequence-pair classification task. An FNet
        sequence pair mask has the following format:

        ```
        0 0 0 0 0 0 0 0 0 0 0 1 1 1 1 1 1 1 1 1
        | first sequence    | second sequence |
        ```

        if token_ids_1 is None, only returns the first portion of the mask (0s).

        Args:
            token_ids_0 (`List[int]`):
                List of ids.
            token_ids_1 (`List[int]`, *optional*):
                Optional second list of IDs for sequence pairs.

        Returns:
            `List[int]`: List of [token type IDs](../glossary#token-type-ids) according to the given sequence(s).
        r   r   )r1   r2   Úlenr3   s        r&   Ú$create_token_type_ids_from_sequencesz6FNetTokenizerFast.create_token_type_ids_from_sequences‘   st   € ð. × Ñ Ð!ˆØ× Ñ Ð!ˆàÐÜ�s˜[Ñ(¨3Ñ.Ó/°1°#Ñ5Ð5Ü�3˜Ñ$ sÑ*Ó+¨q¨cÑ1´C¸ÀcÑ8IÓ4JÈaÈSÑ4PÑPÐPr'   Úsave_directoryÚfilename_prefixc                 óš  — t         j                  j                  |«      st        j	                  d|› d�«       y t         j                  j                  ||r|dz   ndt        d   z   «      }t         j                  j                  | j                  «      t         j                  j                  |«      k7  rt        | j                  |«       |fS )NzVocabulary path (z) should be a directoryú-Ú r   )
r*   r+   ÚisdirÚloggerÚerrorÚjoinÚVOCAB_FILES_NAMESÚabspathr   r   )r#   r:   r;   Úout_vocab_files       r&   Úsave_vocabularyz!FNetTokenizerFast.save_vocabulary¯   s˜   € Ü�w‰w�}‰}˜^Ô,Ü�L‰LÐ,¨^Ð,<Ð<SÐTÔUØÜŸ™Ÿ™Ø±o˜_¨sÒ2È2ÔQbÐcoÑQpÑpó
ˆô �7‰7�?‰?˜4Ÿ?™?Ó+¬r¯w©w¯©¸~Ó/NÒNÜ�T—_‘_ nÔ5àÐ Ð r'   )
NNFTTz<unk>z[SEP]z<pad>z[CLS]z[MASK])N)Ú__name__Ú
__module__Ú__qualname__Ú__doc__rC   Úvocab_files_namesÚmodel_input_namesr   Úslow_tokenizer_classr"   ÚpropertyÚboolr-   r   Úintr   r6   r9   r    r   rF   Ú__classcell__)r%   s   @r&   r   r   &   sú   ø„ ñ ðD *ÐØ$Ð&6Ð7ÐØ(Ðð ØØØØØØØØØõ%%ðN ðM¨ò Mó ðMð JNñ;Ø ™9ð;Ø3;¸DÀ¹IÑ3Fð;à	ˆc‰ó;ð4 JNñQØ ™9ðQØ3;¸DÀ¹IÑ3FðQà	ˆc‰óQñ<!¨cð !ÀHÈSÁMð !Ð]bÐcfÑ]g÷ !r'   r   )rJ   r*   Úshutilr   Útypingr   r   r   Útokenization_utilsr   Útokenization_utils_fastr	   Úutilsr
   r   Útokenization_fnetr   Ú
get_loggerrG   r@   rC   ÚSPIECE_UNDERLINEr   Ú__all__© r'   r&   ú<module>r\      so   ðñ +ã 	Ý ß (Ñ (å ,Ý >ß 8ñ ÔÞ0à€Mà	ˆ×	Ñ	˜HÓ	%€Ø#1ÐEUÑVÐ ð Ð ôT!Ð/ô T!ðn Ð
�r'   