Ë
    S^(h  ã                   ó¦   — d dl mZmZmZmZ ddlmZ ddlmZ ddl	m
Z
mZmZ ddlmZmZ ddlmZ dd	lmZ  G d
„ de
d¬«      Z G d„ de«      ZdgZy)é    )ÚDictÚListÚOptionalÚUnioné   )ÚBatchFeature)Ú
ImageInput)ÚProcessingKwargsÚProcessorMixinÚUnpack)ÚPreTokenizedInputÚ	TextInput)Ú
TensorTypeé   )ÚAutoTokenizerc                   ó6   — e Zd Zddidddœej                  dœZy)ÚAriaProcessorKwargsÚpaddingFéÔ  )Úmax_image_sizeÚsplit_image)Útext_kwargsÚimages_kwargsÚreturn_tensorsN)Ú__name__Ú
__module__Ú__qualname__r   ÚPYTORCHÚ	_defaults© ó    úf/var/www/skyplay_api_hub/venv/lib/python3.12/site-packages/transformers/models/aria/processing_aria.pyr   r      s-   „ ð �uð
ð "Ø ñ
ð %×,Ñ,ñ	�Ir!   r   F)Útotalc                   óÞ   ‡ — e Zd ZdZddgZddgZdZdZ	 	 	 	 ddee	e
f   dee
   deeeeef   ef      fˆ fd„Z	 	 	 dd	eeeee   ee   f   d
ee   dee   defd„Zd„ Zd„ Zed„ «       Zˆ xZS )ÚAriaProcessoraß  
    AriaProcessor is a processor for the Aria model which wraps the Aria image preprocessor and the LLama slow tokenizer.

    Args:
        image_processor (`AriaImageProcessor`, *optional*):
            The AriaImageProcessor to use for image preprocessing.
        tokenizer (`PreTrainedTokenizerBase`, *optional*):
            An instance of [`PreTrainedTokenizerBase`]. This should correspond with the model's text model. The tokenizer is a required input.
        chat_template (`str`, *optional*):
            A Jinja template which will be used to convert lists of messages in a chat into a tokenizable string.
        size_conversion (`Dict`, *optional*):
            A dictionary indicating size conversions for images.
    Úimage_processorÚ	tokenizerÚchat_templateÚsize_conversionÚAriaImageProcessorr   c                 óæ   •— |€dddœ}|j                  «       D ��ci c]  \  }}t        |«      |“Œ c}}| _        |�|j                  €|j                  |_        t
        ‰| �  |||¬«       y c c}}w )Né€   é   )iê  r   )r(   )ÚitemsÚintr)   Ú	pad_tokenÚ	unk_tokenÚsuperÚ__init__)Úselfr&   r'   r(   r)   ÚkÚvÚ	__class__s          €r"   r3   zAriaProcessor.__init__@   sv   ø€ ð Ð"Ø$'¨cÑ2ˆOØ6E×6KÑ6KÓ6M×N©d¨a°¤ A£¨¡	ÓNˆÔàÐ  Y×%8Ñ%8Ð%@Ø"+×"5Ñ"5ˆIÔä‰Ñ˜¨)À=ÐÕQùó  Os   œA-ÚtextÚimagesÚkwargsÚreturnc                 óˆ  —  | j                   t        fd| j                  j                  i|¤Ž}t	        |t
        «      r|g}n.t	        |t        «      st	        |d   t
        «      st        d«      ‚|�¨ | j                  |fi |d   ¤Ž}| j                  |j                  j                  d      }g }	|j                  d«      |z  }
|D ]P  }|j                  | j                  j                  | j                  j                  |
z  «      }|	j                  |«       ŒR ni }|}	 | j                  |	fi |d   ¤Ž}t!        i |¥|¥¬«      S )	a¡  
        Main method to prepare for the model one or several sequences(s) and image(s).

        Args:
            text (`TextInput`, `PreTokenizedInput`, `List[TextInput]`, `List[PreTokenizedInput]`):
                The sequence or batch of sequences to be encoded. Each sequence can be a string or a list of strings
                (pretokenized string). If the sequences are provided as list of strings (pretokenized), you must set
                `is_split_into_words=True` (to lift the ambiguity with a batch of sequences).
            images (`ImageInput`):
                The image or batch of images to be prepared. Each image can be a PIL image, NumPy array or PyTorch
                tensor. Both channels-first and channels-last formats are supported.


        Returns:
            [`BatchFeature`]: A [`BatchFeature`] with the following fields:
            - **input_ids** -- List of token ids to be fed to a model. Returned when `text` is not `None`.
            - **attention_mask** -- List of indices specifying which tokens should be attended to by the model (when
            `return_attention_mask=True` or if *"attention_mask"* is in `self.model_input_names` and if `text` is not
            `None`).
            - **pixel_values** -- Pixel values to be fed to a model. Returned when `images` is not `None`.
            - **pixel_mask** -- Pixel mask to be fed to a model. Returned when `images` is not `None`.
        Útokenizer_init_kwargsr   zAInvalid input text. Please provide a string, or a list of stringsr   r   Ú	num_cropsr   )Údata)Ú_merge_kwargsr   r'   Úinit_kwargsÚ
isinstanceÚstrÚlistÚ
ValueErrorr&   r)   Úpixel_valuesÚshapeÚpopÚreplaceÚimage_tokenÚappendr   )r4   r8   r9   ÚaudioÚvideosr:   Úoutput_kwargsÚimage_inputsÚtokens_per_imageÚprompt_stringsr>   ÚsampleÚtext_inputss                r"   Ú__call__zAriaProcessor.__call__P   s^  € ð< +˜×*Ñ*Üñ
à"&§.¡.×"<Ñ"<ð
ð ñ
ˆô
 �dœCÔ Ø�6‰DÜ˜D¤$Ô'´
¸4À¹7ÄCÔ0HÜÐ`ÓaÐaØÐØ/˜4×/Ñ/Øñà Ñ0ñˆLð
  $×3Ñ3°L×4MÑ4M×4SÑ4SÐTUÑ4VÑWÐØˆNØ$×(Ñ(¨Ó5Ð8HÑHˆIØò .�ØŸ™¨¯©×(BÑ(BÀDÇNÁN×D^ÑD^ÐajÑDjÓk�Ø×%Ñ% fÕ-ñ.ð
 ˆLØ!ˆNà$�d—n‘nØñ
à˜MÑ*ñ
ˆô
 Ð!@ KÐ!@°<Ð!@ÔAÐAr!   c                 ó:   —  | j                   j                  |i |¤ŽS )zÂ
        This method forwards all its arguments to LlamaTokenizerFast's [`~PreTrainedTokenizer.batch_decode`]. Please
        refer to the docstring of this method for more information.
        )r'   Úbatch_decode©r4   Úargsr:   s      r"   rV   zAriaProcessor.batch_decode�   s    € ð
 +ˆt�~‰~×*Ñ*¨DÐ;°FÑ;Ð;r!   c                 ó:   —  | j                   j                  |i |¤ŽS )z¼
        This method forwards all its arguments to LlamaTokenizerFast's [`~PreTrainedTokenizer.decode`]. Please refer to
        the docstring of this method for more information.
        )r'   ÚdecoderW   s      r"   rZ   zAriaProcessor.decode–   s    € ð
 %ˆt�~‰~×$Ñ$ dÐ5¨fÑ5Ð5r!   c                 óÐ   — | j                   j                  }| j                  j                  }|D �cg c]
  }|dk7  sŒ	|‘Œ }}t        t        j                  ||z   «      «      S c c}w )Nr>   )r'   Úmodel_input_namesr&   rD   ÚdictÚfromkeys)r4   Útokenizer_input_namesÚimage_processor_input_namesÚnames       r"   r\   zAriaProcessor.model_input_names�   se   € à $§¡× @Ñ @ÐØ&*×&:Ñ&:×&LÑ&LÐ#ð 9TÖ&k°ÐW[Ð_jÓWj¢tÐ&kÐ#Ð&kÜ”D—M‘MÐ"7Ð:UÑ"UÓVÓWÐWùò 'ls
   ±
A#¼A#)NNNN)NNN)r   r   r   Ú__doc__Ú
attributesÚvalid_kwargsÚimage_processor_classÚtokenizer_classr   r   rC   r   r   Úfloatr/   r3   r   r   r   r	   r   r   r   rT   rV   rZ   Úpropertyr\   Ú__classcell__)r7   s   @r"   r%   r%   ,   s  ø„ ñð $ [Ð1€JØ#Ð%6Ð7€LØ0ÐØ%€Oð Ø/3Ø'+ØBFñRð ˜¨Ð+Ñ,ðRð   ‘}ð	Rð
 " $ u¨U°C¨ZÑ'8¸#Ð'=Ñ">Ñ?õRð& (,ØØñ=Bà�IÐ0°$°y±/À4ÐHYÑCZÐZÑ[ð=Bð ˜Ñ$ð=Bð Ð,Ñ-ð=Bð 
ó=Bò~<ò6ð ñXó ôXr!   r%   N)Útypingr   r   r   r   Úimage_processing_utilsr   Úimage_utilsr	   Úprocessing_utilsr
   r   r   Útokenization_utilsr   r   Úutilsr   Úautor   r   r%   Ú__all__r    r!   r"   ú<module>rr      sL   ð÷* /Ó .å 2Ý %ß HÑ Hß >Ý Ý  ô
Ð*°%õ 
ôyX�Nô yXðx Ð
�r!   