Ë
    T^(hp>  ã                   óì   — d Z ddlZddlZddlZddlZddlmZ ddlmZm	Z	m
Z
mZmZmZ ddlZddlmZ ddlmZ ddlmZ erdd	lmZ dd
lmZmZ  ej4                  e«      ZddiZdZ G d„ de«      ZdgZ y)z$Tokenization class for SigLIP model.é    N)Úcopyfile)ÚTYPE_CHECKINGÚAnyÚDictÚListÚOptionalÚTupleé   )Úimport_protobuf)ÚPreTrainedTokenizer)Ú
AddedToken)Ú	TextInput)ÚloggingÚrequires_backendsÚ
vocab_filezspiece.modelu   â–�c            
       ó¶  ‡ — e Zd ZdZeZddgZ	 	 	 	 	 	 	 d#deee	e
f      ddfˆ fd„Zd„ Zed	„ «       Zd
„ Z	 d$dee   deee      dedee   fˆ fd„Zdee   dee   fd„Z	 d%dee   deee      dee   fd„Z	 d%dee   deee      dee   fd„Zd„ Zd„ Zde	de	fd„Zddœd„Zd&dddee	   fˆ fd„Zed„ «       Zd„ Zd„ Zd„ Zd„ Z d%d e	d!ee	   de!e	   fd"„Z"ˆ xZ#S )'ÚSiglipTokenizeraè  
    Construct a Siglip tokenizer. Based on [SentencePiece](https://github.com/google/sentencepiece).

    This tokenizer inherits from [`PreTrainedTokenizer`] which contains most of the main methods. Users should refer to
    this superclass for more information regarding those methods.

    Args:
        vocab_file (`str`):
            [SentencePiece](https://github.com/google/sentencepiece) file (generally has a *.spm* extension) that
            contains the vocabulary necessary to instantiate a tokenizer.
        eos_token (`str`, *optional*, defaults to `"</s>"`):
            The end of sequence token.
        unk_token (`str`, *optional*, defaults to `"<unk>"`):
            The unknown token. A token that is not in the vocabulary cannot be converted to an ID and is set to be this
            token instead.
        pad_token (`str`, *optional*, defaults to `"</s>"`):
            The token used for padding, for example when batching sequences of different lengths.
        additional_special_tokens (`List[str]`, *optional*):
            Additional special tokens used by the tokenizer.
        sp_model_kwargs (`dict`, *optional*):
            Will be passed to the `SentencePieceProcessor.__init__()` method. The [Python wrapper for
            SentencePiece](https://github.com/google/sentencepiece/tree/master/python) can be used, among other things,
            to set:

            - `enable_sampling`: Enable subword regularization.
            - `nbest_size`: Sampling parameters for unigram. Invalid for BPE-Dropout.

              - `nbest_size = {0,1}`: No sampling is performed.
              - `nbest_size > 1`: samples from the nbest_size results.
              - `nbest_size < 0`: assuming that nbest_size is infinite and samples from the all hypothesis (lattice)
                using forward-filtering-and-backward-sampling algorithm.

            - `alpha`: Smoothing parameter for unigram sampling, and dropout probability of merge operations for
              BPE-dropout.
        model_max_length (`int`, *optional*, defaults to 64):
            The maximum length (in number of tokens) for model inputs.
        do_lower_case (`bool`, *optional*, defaults to `True`):
            Whether or not to lowercase the input when tokenizing.
    Ú	input_idsÚattention_maskNÚsp_model_kwargsÚreturnc	                 ó–  •— t        | d«       t        |t        «      rt        |dddd¬«      n|}t        |t        «      rt        |dddd¬«      n|}t        |t        «      rt        |dddd¬«      n|}|€i n|| _        || _        || _        | j                  «       | _        || _        t        ‰
| �(  d||||| j                  ||dœ|	¤Ž y )NÚprotobufTF)ÚrstripÚlstripÚ
normalizedÚspecial)Ú	eos_tokenÚ	unk_tokenÚ	pad_tokenÚadditional_special_tokensr   Úmodel_max_lengthÚdo_lower_case© )r   Ú
isinstanceÚstrr   r   r#   r   Úget_spm_processorÚsp_modelÚsuperÚ__init__)Úselfr   r   r   r    r!   r   r"   r#   ÚkwargsÚ	__class__s             €úl/var/www/skyplay_api_hub/venv/lib/python3.12/site-packages/transformers/models/siglip/tokenization_siglip.pyr*   zSiglipTokenizer.__init__X   sò   ø€ ô 	˜$ 
Ô+ô ˜)¤SÔ)ô �y¨°dÀuÐVZÕ[àð 	ô ˜)¤SÔ)ô �y¨°dÀuÐVZÕ[àð 	ô ˜)¤SÔ)ô �y¨°dÀuÐVZÕ[àð 	ð &5Ð%<™rÀ/ˆÔà*ˆÔØ$ˆŒà×.Ñ.Ó0ˆŒØ$ˆŒä‰Ñð 		
ØØØØ&?Ø ×0Ñ0Ø-Ø'ñ		
ð ó		
ó    c                 ó¬  — t        j                  di | j                  ¤Ž}t        | j                  d«      5 }|j                  «       }t        «       }|j                  j                  |«      }|j                  «       }d|_
        |j                  j                  |«       |j                  «       }|j                  |«       d d d «       |S # 1 sw Y   |S xY w)NÚrbFr$   )ÚspmÚSentencePieceProcessorr   Úopenr   Úreadr   Ú
ModelProtoÚ
FromStringÚNormalizerSpecÚadd_dummy_prefixÚnormalizer_specÚ	MergeFromÚSerializeToStringÚLoadFromSerializedProto)r+   Ú	tokenizerÚfr(   Ú	model_pb2Úmodelr:   s          r.   r'   z!SiglipTokenizer.get_spm_processor‰   sº   € Ü×.Ñ.ÑF°×1EÑ1EÑFˆ	Ü�$—/‘/ 4Ó(ð 	8¨AØ—v‘v“xˆHÜ'Ó)ˆIØ×(Ñ(×3Ñ3°HÓ=ˆEØ'×6Ñ6Ó8ˆOØ/4ˆOÔ,Ø×!Ñ!×+Ñ+¨OÔ<Ø×.Ñ.Ó0ˆHØ×-Ñ-¨hÔ7÷	8ð Ð÷	8ð Ðús   ¶B	C	Ã	Cc                 ó6   — | j                   j                  «       S ©N)r(   Úget_piece_size©r+   s    r.   Ú
vocab_sizezSiglipTokenizer.vocab_size–   s   € ð �}‰}×+Ñ+Ó-Ð-r/   c                 óª   — t        | j                  «      D �ci c]  }| j                  |«      |“Œ }}|j                  | j                  «       |S c c}w rC   )ÚrangerF   Úconvert_ids_to_tokensÚupdateÚadded_tokens_encoder)r+   ÚiÚvocabs      r.   Ú	get_vocabzSiglipTokenizer.get_vocabœ   sK   € Ü;@ÀÇÁÓ;QÖR°a�×+Ñ+¨AÓ.°Ñ1ÐRˆÐRØ�‰�T×.Ñ.Ô/Øˆùò Ss   ˜AÚtoken_ids_0Útoken_ids_1Úalready_has_special_tokensc                 ó¤   •— |rt         ‰| �  ||d¬«      S |€dgt        |«      z  dgz   S dgt        |«      z  dgz   dgt        |«      z  z   dgz   S )aÄ  
        Retrieve sequence ids from a token list that has no special tokens added. This method is called when adding
        special tokens using the tokenizer `prepare_for_model` method.

        Args:
            token_ids_0 (`List[int]`):
                List of IDs.
            token_ids_1 (`List[int]`, *optional*):
                Optional second list of IDs for sequence pairs.
            already_has_special_tokens (`bool`, *optional*, defaults to `False`):
                Whether or not the token list is already formatted with special tokens for the model.

        Returns:
            `List[int]`: A list of integers in the range [0, 1]: 1 for a special token, 0 for a sequence token.
        T)rO   rP   rQ   r   é   )r)   Úget_special_tokens_maskÚlen)r+   rO   rP   rQ   r-   s       €r.   rT   z'SiglipTokenizer.get_special_tokens_mask¢   sy   ø€ ñ$ &Ü‘7Ñ2Ø'°[Ð]að 3ó ð ð
 ÐØ�Cœ#˜kÓ*Ñ*¨q¨cÑ1Ð1Ø�”c˜+Ó&Ñ&¨1¨#Ñ-°!°´s¸;Ó7GÑ1GÑHÈAÈ3ÑNÐNr/   Ú	token_idsc                 ó¬   — t        |«      dkD  r7|d   | j                  k(  r%t        j                  d| j                  › d�«       |S || j                  gz   S )z.Do not add eos again if user already added it.r   éÿÿÿÿzThis sequence already has zQ. In future versions this behavior may lead to duplicated eos tokens being added.)rU   Úeos_token_idÚwarningsÚwarnr   )r+   rV   s     r.   Ú_add_eos_if_not_presentz'SiglipTokenizer._add_eos_if_not_present¿   s]   € äˆy‹>˜AÒ )¨B¡-°4×3DÑ3DÒ"DÜ�M‰MØ,¨T¯^©^Ð,<ð =+ð +ôð Ðà × 1Ñ 1Ð2Ñ2Ð2r/   c                 ót   — | j                   g}|€t        ||z   «      dgz  S t        ||z   |z   |z   «      dgz  S )aÇ  
        Create a mask from the two sequences passed to be used in a sequence-pair classification task. T5 does not make
        use of token type ids, therefore a list of zeros is returned.

        Args:
            token_ids_0 (`List[int]`):
                List of IDs.
            token_ids_1 (`List[int]`, *optional*):
                Optional second list of IDs for sequence pairs.

        Returns:
            `List[int]`: List of zeros.
        r   )rY   rU   )r+   rO   rP   Úeoss       r.   Ú$create_token_type_ids_from_sequencesz4SiglipTokenizer.create_token_type_ids_from_sequencesË   sP   € ð  × Ñ Ð!ˆàÐÜ�{ SÑ(Ó)¨Q¨CÑ/Ð/Ü�; Ñ$ {Ñ2°SÑ8Ó9¸Q¸CÑ?Ð?r/   c                 óX   — | j                  |«      }|€|S | j                  |«      }||z   S )a‚  
        Build model inputs from a sequence or a pair of sequence for sequence classification tasks by concatenating and
        adding special tokens. A sequence has the following format:

        - single sequence: `X </s>`
        - pair of sequences: `A </s> B </s>`

        Args:
            token_ids_0 (`List[int]`):
                List of IDs to which the special tokens will be added.
            token_ids_1 (`List[int]`, *optional*):
                Optional second list of IDs for sequence pairs.

        Returns:
            `List[int]`: List of [input IDs](../glossary#input-ids) with the appropriate special tokens.
        )r\   )r+   rO   rP   s      r.   Ú build_inputs_with_special_tokensz0SiglipTokenizer.build_inputs_with_special_tokensâ   s;   € ð& ×2Ñ2°;Ó?ˆØÐØÐà×6Ñ6°{ÓCˆKØ Ñ,Ð,r/   c                 óD   — | j                   j                  «       }d |d<   |S )Nr(   )Ú__dict__Úcopy)r+   Ústates     r.   Ú__getstate__zSiglipTokenizer.__getstate__ý   s#   € Ø—‘×"Ñ"Ó$ˆØ ˆˆjÑØˆr/   c                 óÊ   — || _         t        | d«      si | _        t        j                  di | j                  ¤Ž| _        | j
                  j                  | j                  «       y )Nr   r$   )rc   Úhasattrr   r2   r3   r(   ÚLoadr   )r+   Úds     r.   Ú__setstate__zSiglipTokenizer.__setstate__  sO   € ØˆŒô �tÐ.Ô/Ø#%ˆDÔ ä×2Ñ2ÑJ°T×5IÑ5IÑJˆŒØ�‰×Ñ˜4Ÿ?™?Õ+r/   Útextc                 ój   — |j                  t        j                  ddt        j                  «      «      S )NÚ )Ú	translater&   Ú	maketransÚstringÚpunctuation)r+   rl   s     r.   Úremove_punctuationz"SiglipTokenizer.remove_punctuation  s$   € Ø�~‰~œcŸm™m¨B°´F×4FÑ4FÓGÓHÐHr/   ©Úkeep_punctuation_exact_stringc                óÐ   ‡ — |r*|j                  ˆ fd„|j                  |«      D «       «      }n‰ j                  |«      }t        j                  dd|«      }|j                  «       }|S )a•  Returns canonicalized `text` (puncuation removed).

        Args:
            text (`str`):
                String to be canonicalized.
            keep_punctuation_exact_string (`str`, *optional*):
                If provided, then this exact string is kept. For example providing '{}' will keep any occurrences of '{}'
                (but will still remove '{' and '}' that appear separately).
        c              3   ó@   •K  — | ]  }‰j                  |«      –— Œ y ­wrC   )rs   )Ú.0Úpartr+   s     €r.   ú	<genexpr>z4SiglipTokenizer.canonicalize_text.<locals>.<genexpr>  s!   øè ø€ ò 6Ø26�×'Ñ'¨×-ñ6ùs   ƒz\s+ú )ÚjoinÚsplitrs   ÚreÚsubÚstrip)r+   rl   ru   s   `  r.   Úcanonicalize_textz!SiglipTokenizer.canonicalize_text  sc   ø€ ñ )Ø0×5Ñ5ó 6Ø:>¿*¹*ÐEbÓ:cô6ó ‰Dð ×*Ñ*¨4Ó0ˆDÜ�v‰v�f˜c 4Ó(ˆØ�z‰z‹|ˆàˆr/   r   c                 ó¾   •— t        ‰| �  t        |j                  t        d«      z   fi |¤Ž}t	        |«      dkD  r"|d   t        k(  r|d   | j
                  v r|dd }|S )z8
        Converts a string to a list of tokens.
        r{   rS   r   N)r)   ÚtokenizeÚSPIECE_UNDERLINEÚreplacerU   Úall_special_tokens)r+   rl   Úadd_special_tokensr,   Útokensr-   s        €r.   rƒ   zSiglipTokenizer.tokenize&  se   ø€ ô ‘Ñ!Ô"2°T·\±\ÔBRÐTWÓ5XÑ"XÑcÐ\bÑcˆäˆv‹;˜Š?˜v a™yÔ,<Ò<ÀÈÁÈd×NeÑNeÑAeØ˜A˜B�ZˆFØˆr/   c                 óp   — t        | j                  j                  t        | j                  «      «      «      S rC   )rU   r(   Úencoder&   r   rE   s    r.   Úunk_token_lengthz SiglipTokenizer.unk_token_length0  s'   € ô �4—=‘=×'Ñ'¬¨D¯N©NÓ(;Ó<Ó=Ð=r/   c                 ó  — | j                  |d¬«      }| j                  j                  |t        ¬«      }| j                  j                  | j                  |z   t        ¬«      }t        |«      | j                  k\  r|| j                  d S |S )u*  
        Returns a tokenized string.

        We de-activated the `add_dummy_prefix` option, thus the sentencepiece internals will always strip any
        SPIECE_UNDERLINE.

        For example: `self.sp_model.encode(f"{SPIECE_UNDERLINE}Hey", out_type = str)` will give `['H', 'e', 'y']` instead of `['â–�He', 'y']`.

        Thus we always encode `f"{unk_token}text"` and strip the `unk_token`. Here is an example with `unk_token = "<unk>"` and `unk_token_length = 4`.
        `self.tokenizer.sp_model.encode("<unk> Hey", out_type = str)[4:]`.
        Nrt   )Úout_type)r�   r(   rŠ   r&   r   rU   r‹   )r+   rl   r,   rˆ   s       r.   Ú	_tokenizezSiglipTokenizer._tokenize5  s�   € ð ×%Ñ% dÈ$Ð%ÓOˆØ—‘×%Ñ% d´SÐ%Ó9ˆð —‘×%Ñ% d§n¡n°tÑ&;ÄcÐ%ÓJˆä25°f³+À×AVÑAVÒ2Vˆv�d×+Ñ+Ð-Ð.ÐbÐ\bÐbr/   c                 ó8   — | j                   j                  |«      S )z0Converts a token (str) in an id using the vocab.)r(   Úpiece_to_id)r+   Útokens     r.   Ú_convert_token_to_idz$SiglipTokenizer._convert_token_to_idJ  s   € à�}‰}×(Ñ(¨Ó/Ð/r/   c                 ó<   — | j                   j                  |«      }|S )z=Converts an index (integer) in a token (str) using the vocab.)r(   Ú	IdToPiece)r+   Úindexr‘   s      r.   Ú_convert_id_to_tokenz$SiglipTokenizer._convert_id_to_tokenO  s   € à—‘×'Ñ'¨Ó.ˆØˆr/   c                 ó  — g }d}d}|D ]P  }|| j                   v r-|s|dz  }|| j                  j                  |«      |z   z  }d}g }Œ>|j                  |«       d}ŒR || j                  j                  |«      z  }|j	                  «       S )z:Converts a sequence of tokens (string) in a single string.rn   Fr{   T)r†   r(   ÚdecodeÚappendr€   )r+   rˆ   Úcurrent_sub_tokensÚ
out_stringÚprev_is_specialr‘   s         r.   Úconvert_tokens_to_stringz(SiglipTokenizer.convert_tokens_to_stringT  s¤   € àÐØˆ
ØˆØò 
	(ˆEà˜×/Ñ/Ñ/Ù&Ø #Ñ%�JØ˜dŸm™m×2Ñ2Ð3EÓFÈÑNÑN�
Ø"&�Ø%'Ñ"à"×)Ñ)¨%Ô0Ø"'‘ð
	(ð 	�d—m‘m×*Ñ*Ð+=Ó>Ñ>ˆ
Ø×ÑÓ!Ð!r/   Úsave_directoryÚfilename_prefixc                 óæ  — t         j                  j                  |«      st        j	                  d|› d�«       y t         j                  j                  ||r|dz   ndt        d   z   «      }t         j                  j                  | j                  «      t         j                  j                  |«      k7  rBt         j                  j                  | j                  «      rt        | j                  |«       |fS t         j                  j                  | j                  «      sCt        |d«      5 }| j                  j                  «       }|j                  |«       d d d «       |fS |fS # 1 sw Y   |fS xY w)NzVocabulary path (z) should be a directoryú-rn   r   Úwb)ÚosÚpathÚisdirÚloggerÚerrorr|   ÚVOCAB_FILES_NAMESÚabspathr   Úisfiler   r4   r(   Úserialized_model_protoÚwrite)r+   rž   rŸ   Úout_vocab_fileÚfiÚcontent_spiece_models         r.   Úsave_vocabularyzSiglipTokenizer.save_vocabularyh  s%  € Ü�w‰w�}‰}˜^Ô,Ü�L‰LÐ,¨^Ð,<Ð<SÐTÔUØÜŸ™Ÿ™Ø±o˜_¨sÒ2È2ÔQbÐcoÑQpÑpó
ˆô �7‰7�?‰?˜4Ÿ?™?Ó+¬r¯w©w¯©¸~Ó/NÒNÔSU×SZÑSZ×SaÑSaÐbf×bqÑbqÔSrÜ�T—_‘_ nÔ5ð Ð Ð ô —‘—‘ §¡Ô0Ü�n dÓ+ð /¨rØ'+§}¡}×'KÑ'KÓ'MÐ$Ø—‘Ð-Ô.÷/ð Ð Ð �Ð Ð ÷	/ð Ð Ð ús   Ä+,E%Å%E0)ú</s>z<unk>r±   NNé@   T)NFrC   )F)$Ú__name__Ú
__module__Ú__qualname__Ú__doc__r¨   Úvocab_files_namesÚmodel_input_namesr   r   r&   r   r*   r'   ÚpropertyrF   rN   r   ÚintÚboolrT   r\   r_   ra   rf   rk   rs   r�   rƒ   r‹   rŽ   r’   r–   r�   r	   r°   Ú__classcell__)r-   s   @r.   r   r   ,   sØ  ø„ ñ&ðP *ÐØ$Ð&6Ð7Ðð
 ØØØ"&Ø48ØØñ/
ð " $ s¨C x¡.Ñ1ð/
ð 
õ/
òbð ñ.ó ð.òð sxñOØ ™9ðOØ3;¸DÀ¹IÑ3FðOØkoðOà	ˆc‰õOð:	3°°c±ð 	3¸tÀC¹yó 	3ð JNñ@Ø ™9ð@Ø3;¸DÀ¹IÑ3Fð@à	ˆc‰ó@ð0 JNñ-Ø ™9ð-Ø3;¸DÀ¹IÑ3Fð-à	ˆc‰ó-ò6ò,ðI sð I¨só Ið HLô ñ*˜[ð ÐQUÐVYÑQZõ ð ñ>ó ð>òcò*0ò
ò
"ñ(!¨cð !ÀHÈSÁMð !Ð]bÐcfÑ]g÷ !r/   r   )!r¶   r£   r~   rq   rZ   Úshutilr   Útypingr   r   r   r   r   r	   Úsentencepiecer2   Úconvert_slow_tokenizerr   Útokenization_utilsr   Útokenization_utils_baser   r   Úutilsr   r   Ú
get_loggerr³   r¦   r¨   r„   r   Ú__all__r$   r/   r.   ú<module>rÆ      sw   ðñ +ã 	Û 	Û Û Ý ß B× Bã å 5Ý 5Ý 1ñ Ý4ß /ð 
ˆ×	Ñ	˜HÓ	%€à! >Ð2Ð ð Ð ôK!Ð)ô K!ð\
 Ð
�r/   