Ë
    T^(h/E  ã            
       óD  — d Z ddlmZmZmZ ddlZddlmZ ddl	m
Z
mZ ddlmZmZmZmZ ddlmZmZ  G d	„ d
ed¬«      Z G d„ ded¬«      Zdee   dedeee      fd„Zdeeee         deee      dededej0                  f
d„Zdedededefd„Z G d„ de«      ZdgZy)zProcessor class for Mllama.é    )ÚListÚOptionalÚUnionNé   )ÚBatchFeature)Ú
ImageInputÚmake_nested_list_of_images)ÚImagesKwargsÚProcessingKwargsÚProcessorMixinÚUnpack)ÚPreTokenizedInputÚ	TextInputc                   ó   — e Zd ZU ee   ed<   y)ÚMllamaImagesKwargsÚmax_image_tilesN)Ú__name__Ú
__module__Ú__qualname__r   ÚintÚ__annotations__© ó    új/var/www/skyplay_api_hub/venv/lib/python3.12/site-packages/transformers/models/mllama/processing_mllama.pyr   r      s   … Ø˜c‘]Ô"r   r   F)Útotalc                   ó$   — e Zd ZU eed<   dddiiZy)ÚMllamaProcessorKwargsÚimages_kwargsÚimage_kwargsr   é   N)r   r   r   r   r   Ú	_defaultsr   r   r   r   r   #   s   … Ø%Ó%ð 	Ø˜qð
ð�Ir   r   Ú	input_idsÚimage_token_idÚreturnc                 ó”  — t        | «      D ��cg c]  \  }}||k(  sŒ|‘Œ }}}t        |«      dk(  rg S t        |«      dk(  r|d   dggS t        |dd |dd «      D ��cg c]	  \  }}||g‘Œ }}}|j                  |d   t        | «      g«       |d   d   }|ddd…   D ]  }	|	d   |	d   dz
  k(  r||	d<   |	d   }Œ |S c c}}w c c}}w )aó  
    Generate a cross-attention token mask for image tokens in the input sequence.

    This function identifies the positions of image tokens in the input sequence and creates
    a mask that defines which subsequent tokens each image token should attend to.

    Args:
        input_ids (List[int]): A list of token ids representing the input sequence.
        image_token_id (int): The id of the token used to represent images in the sequence.

    Returns:
        List[List[int]]: A list of [start, end] pairs, where each pair represents the range
        of tokens an image token should attend to.

    Notes:
        - If no image tokens are present, an empty list is returned.
        - For a single image token, it attends to all subsequent tokens until the end of the sequence.
        - For multiple image tokens, each attends to tokens up to the next image token or the end of the sequence.
        - Consecutive image tokens are treated as a group and attend to all subsequent tokens together.
    r   é   éÿÿÿÿN)Ú	enumerateÚlenÚzipÚappend)
r"   r#   ÚiÚtokenÚimage_token_locationsÚloc1Úloc2Úvision_masksÚlast_mask_endÚvision_masks
             r   Úget_cross_attention_token_maskr4   -   s  € ô, 09¸Ó/C×_¡8 1 eÀuÐP^ÓG^šQÐ_ÐÑ_ä
Ð Ó! QÒ&Øˆ	ô Ð Ó! QÒ&Ø& qÑ)¨2Ð.Ð/Ð/ä36Ð7LÈSÈbÐ7QÐShÐijÐikÐSlÓ3m×n¡Z T¨4�T˜4’LÐn€LÑnð ×ÑÐ.¨rÑ2´C¸	³NÐCÔDð
 ! Ñ$ QÑ'€MØ#¡D b DÑ)ò 'ˆØ�q‰>˜[¨™^¨aÑ/Ò/Ø*ˆK˜‰NØ# A™‰ð'ð
 Ðùó/ `ùó os   �B>�B>ÁCÚcross_attention_token_maskÚ	num_tilesÚmax_num_tilesÚlengthc           	      ó¤  — t        | «      }t        | D �cg c]  }t        |«      ‘Œ c}«      }t        j                  ||||ft        j                  ¬«      }t        t        | |«      «      D ]\  \  }\  }	}
t        t        |	|
«      «      D ]<  \  }\  }}t        |«      dk(  sŒ|\  }}t        ||«      }|dk(  r|}d||||…|d|…f<   Œ> Œ^ |S c c}w )a  
    Convert the cross attention mask indices to a cross attention mask 4D array.

    This function takes a sparse representation of cross attention masks and converts it to a dense 4D numpy array.
    The sparse representation is a nested list structure that defines attention ranges for each image in each batch item.

    Args:
        cross_attention_token_mask (List[List[List[int]]]): A nested list structure where:
            - The outer list represents the batch dimension.
            - The middle list represents different images within each batch item.
            - The inner list contains pairs of integers [start, end] representing token ranges for each image.
        num_tiles (List[List[int]]): A nested list structure specifying the number of tiles for each image in each batch item.
        max_num_tiles (int): The maximum possible number of tiles.
        length (int): The total sequence length of the input.

    Returns:
        np.ndarray: A 4D numpy array of shape (batch_size, length, max_num_images, max_num_tiles)
            The array contains `1` where attention is allowed and `0` where it is not.

    Note:
        - Special handling is done for cases where the end token is -1, which is interpreted as attending to the end of the sequence.
    )ÚshapeÚdtypeé   r'   r&   N)r)   ÚmaxÚnpÚzerosÚint64r(   r*   Úmin)r5   r6   r7   r8   Ú
batch_sizeÚmasksÚmax_num_imagesÚcross_attention_maskÚ
sample_idxÚsample_masksÚsample_num_tilesÚmask_idxÚ	locationsÚmask_num_tilesÚstartÚends                   r   Ú,convert_sparse_cross_attention_mask_to_denserN   ]   s÷   € ô: Ð/Ó0€JÜÐ2LÖM¨œ#˜e�*ÒMÓN€NäŸ8™8Ø˜6 >°=ÐAÜ�h‰hôÐô
 9BÄ#ÐF`ÐbkÓBlÓ8mò [Ñ4ˆ
Ñ4�\Ð#3Ü5>¼sÀ<ÐQaÓ?bÓ5cò 	[Ñ1ˆHÑ1�y .Ü�9‹~ Ó"Ø&‘
��sÜ˜#˜vÓ&�Ø˜"’9Ø �CØYZÐ$ Z°°s°¸HÀoÀ~ÀoÐ%UÒVñ	[ð[ð  Ðùò Ns   •CÚpromptÚ	bos_tokenÚimage_tokenc                 ó”   — || v r| S d}| j                  |«      r%| t        |«      d } |dz  }| j                  |«      rŒ%||z  › |› | › �S )a\  
    Builds a string from the input prompt by adding `bos_token` if not already present.

    Args:
        prompt (`str`):
            The input prompt string.
        bos_token (`str`):
            The beginning of sentence token to be added.
        image_token (`str`):
            The image token used to identify the start of an image sequence.

    Returns:
        str: The modified prompt string with the `bos_token` added if necessary.

    Examples:
        >>> build_string_from_input("Hello world", "<begin_of_text>", "<|image|>")
        '<begin_of_text>Hello world'

        >>> build_string_from_input("<|image|>Hello world", "<begin_of_text>", "<|image|>")
        '<|image|><begin_of_text>Hello world'

        >>> build_string_from_input("<begin_of_text>Hello world", "<begin_of_text>", "<|image|>")
        '<begin_of_text>Hello world'
    r   Nr&   )Ú
startswithr)   )rO   rP   rQ   Únum_image_tokens_on_starts       r   Úbuild_string_from_inputrU   �   sn   € ð4 �FÑØˆà !ÐØ
×
Ñ
˜KÔ
(Øœ˜KÓ(Ð*Ð+ˆØ! QÑ&Ð!ð ×
Ñ
˜KÕ
(ð Ð5Ñ5Ð6°y°kÀ&ÀÐJÐJr   c                   ó®   ‡ — e Zd ZdZddgZdgZdZdZdˆ fd„	Z	 	 	 	 dde	e
   d	e	eeeee   ee   f      d
ee   defd„Zd„ Zd„ Z	 dd„Zed„ «       Zˆ xZS )ÚMllamaProcessora  
    Constructs a Mllama processor which wraps [`MllamaImageProcessor`] and
    [`PretrainedTokenizerFast`] into a single processor that inherits both the image processor and
    tokenizer functionalities. See the [`~MllamaProcessor.__call__`] and [`~OwlViTProcessor.decode`] for more
    information.
    The preferred way of passing kwargs is as a dictionary per modality, see usage example below.
        ```python
        from transformers import MllamaProcessor
        from PIL import Image

        processor = MllamaProcessor.from_pretrained("meta-llama/Llama-3.2-11B-Vision")

        processor(
            images=your_pil_image,
            text=["<|image|>If I had to write a haiku for this one"],
            images_kwargs = {"size": {"height": 448, "width": 448}},
            text_kwargs = {"padding": "right"},
            common_kwargs = {"return_tensors": "pt"},
        )
        ```

    Args:
        image_processor ([`MllamaImageProcessor`]):
            The image processor is a required input.
        tokenizer ([`PreTrainedTokenizer`, `PreTrainedTokenizerFast`]):
            The tokenizer is a required input.
        chat_template (`str`, *optional*): A Jinja template which will be used to convert lists of messages
            in a chat into a tokenizable string.

    Úimage_processorÚ	tokenizerÚchat_templateÚMllamaImageProcessorÚPreTrainedTokenizerFastc                 óF  •— t        |d«      s(d| _        |j                  | j                  «      | _        n"|j                  | _        |j                  | _        d| _        |j                  | j                  «      | _        |j                  | _        t        ‰| �!  |||¬«       y )NrQ   z	<|image|>z<|python_tag|>)rZ   )	ÚhasattrrQ   Úconvert_tokens_to_idsr#   Úpython_tokenÚpython_token_idrP   ÚsuperÚ__init__)ÚselfrX   rY   rZ   Ú	__class__s       €r   rc   zMllamaProcessor.__init__×   sŒ   ø€ Ü�y -Ô0Ø*ˆDÔØ"+×"AÑ"AÀ$×BRÑBRÓ"SˆDÕà(×4Ñ4ˆDÔØ"+×":Ñ":ˆDÔà,ˆÔØ(×>Ñ>¸t×?PÑ?PÓQˆÔØ"×,Ñ,ˆŒÜ‰Ñ˜¨)À=ÐÕQr   ÚimagesÚtextÚkwargsr$   c           
      ó8  — |€|€t        d«      ‚ | j                  t        fd| j                  j                  i|¤Ž}|d   }|d   }|d   }	i }
|�Ót        |t        «      r|g}n3t        |t        t        f«      rt        d„ |D «       «      st        d«      ‚|D �cg c]  }|j                  | j                  «      ‘Œ }}|D �cg c]#  }t        || j                  | j                  «      ‘Œ% }}|j                  d	d«      } | j                  |fi |¤Ž}|
j                  |«       d
g}|�#t!        |«      }|D �cg c]  }t#        |«      ‘Œ }}|�~t%        d„ D «       «      rt        d„ |D «       «      st        d«      ‚t'        |«      d
kD  rA||k7  r<|€t        d«      ‚d}t'        |«      t'        |«      k(  rd}t        d|› d|› d|› �«      ‚|�5 | j(                  |fi |¤Ž}|j                  d«      }|
j                  |«       |�c|�ad   D �cg c]  }t+        || j,                  «      ‘Œ }}t/        || j(                  j0                  t3        d„ |d   D «       «      ¬«      }||
d<   |	j                  dd«      }t5        |
|¬«      }|S c c}w c c}w c c}w c c}w )a&	  
        Main method to prepare text(s) and image(s) to be fed as input to the model. This method forwards the `text`
        arguments to PreTrainedTokenizerFast's [`~PreTrainedTokenizerFast.__call__`] if `text` is not `None` to encode
        the text. To prepare the image(s), this method forwards the `images` arguments to
        MllamaImageProcessor's [`~MllamaImageProcessor.__call__`] if `images` is not `None`. Please refer
        to the docstring of the above two methods for more information.

        Args:
            images (`PIL.Image.Image`, `np.ndarray`, `torch.Tensor`, `List[PIL.Image.Image]`, `List[np.ndarray]`, `List[torch.Tensor]`):
                The image or batch of images to be prepared. Each image can be a PIL image, NumPy array or PyTorch
                tensor. Both channels-first and channels-last formats are supported.
            text (`str`, `List[str]`, `List[List[str]]`):
                The sequence or batch of sequences to be encoded. Each sequence can be a string or a list of strings
                (pretokenized string). If the sequences are provided as list of strings (pretokenized), you must set
                `is_split_into_words=True` (to lift the ambiguity with a batch of sequences).
            return_tensors (`str` or [`~utils.TensorType`], *optional*):
                If set, will return tensors of a particular framework. Acceptable values are:
                    - `'tf'`: Return TensorFlow `tf.constant` objects.
                    - `'pt'`: Return PyTorch `torch.Tensor` objects.
                    - `'np'`: Return NumPy `np.ndarray` objects.
                    - `'jax'`: Return JAX `jnp.ndarray` objects.
        Returns:
            [`BatchFeature`]: A [`BatchFeature`] with the following fields:

            - **input_ids** -- List of token ids to be fed to a model. Returned when `text` is not `None`.
            - **attention_mask** -- List of indices specifying which tokens should be attended to by the model (when
              `return_attention_mask=True` or if *"attention_mask"* is in `self.model_input_names` and if `text` is not
              `None`).
            - **pixel_values** -- Pixel values to be fed to a model. Returned when `images` is not `None`.
            TODO: add aspect_ratio_ids and aspect_ratio_mask and cross_attention_mask
        Nz'You must specify either text or images.Útokenizer_init_kwargsÚtext_kwargsr   Úcommon_kwargsc              3   ó<   K  — | ]  }t        |t        «      –— Œ y ­w©N)Ú
isinstanceÚstr)Ú.0Úts     r   ú	<genexpr>z+MllamaProcessor.__call__.<locals>.<genexpr>  s   è ø€ Ò=_ÐUV¼jÈÌC×>PÑ=_ùs   ‚zAInvalid input text. Please provide a string, or a list of stringsÚpadding_sider   c              3   ó&   K  — | ]	  }|d k(  –— Œ y­w©r   Nr   ©rq   Ú	batch_imgs     r   rs   z+MllamaProcessor.__call__.<locals>.<genexpr>*  s   è ø€ ÒD i�9 •>ÑDùó   ‚c              3   ó&   K  — | ]	  }|d k(  –— Œ y­wrv   r   rw   s     r   rs   z+MllamaProcessor.__call__.<locals>.<genexpr>*  s   è ø€ ò QØ#,�	˜Q•ñQùry   zaIf a batch of text is provided, there should be either no images or at least one image per samplez@No image were provided, but there are image tokens in the promptÚ zZMake sure to pass your images as a nested list, where each sub-list holds images per batchz)The number of image tokens in each text (zA) should be the same as the number of provided images per batch (z). r6   r"   c              3   ó2   K  — | ]  }t        |«      –— Œ y ­wrn   )r)   )rq   r"   s     r   rs   z+MllamaProcessor.__call__.<locals>.<genexpr>J  s   è ø€ ÒQ¨iœ3˜yŸ>ÑQùs   ‚)r6   r7   r8   rE   Úreturn_tensors)ÚdataÚtensor_type)Ú
ValueErrorÚ_merge_kwargsr   rY   Úinit_kwargsro   rp   ÚlistÚtupleÚallÚcountrQ   rU   rP   ÚpopÚupdater	   r)   ÚanyÚsumrX   r4   r#   rN   r   r=   r   )rd   rf   rg   ÚaudioÚvideosrh   Úoutput_kwargsrk   r   rl   r~   rr   Ún_images_in_textÚ	text_itemÚ_ÚencodingÚn_images_in_imagesÚsampleÚadd_messageÚimage_featuresr6   Ú	token_idsr5   rE   r}   Úbatch_features                             r   Ú__call__zMllamaProcessor.__call__ä   s  € ðN ˆ<˜F˜NÜÐFÓGÐGà*˜×*Ñ*Ü!ñ
à"&§.¡.×"<Ñ"<ð
ð ñ
ˆð $ MÑ2ˆØ% oÑ6ˆØ% oÑ6ˆàˆØÐÜ˜$¤Ô$Ø�v‘Ü  ¬¬e }Ô5¼#Ñ=_ÐZ^Ô=_Ô:_Ü Ð!dÓeÐeØCGÖH¸a §¡¨×(8Ñ(8Õ 9ÐHÐÐHØjnÖoÐ]fÔ+¨I°t·~±~Àt×GWÑGWÕXÐoˆDÐoØ—‘ °Ó5ˆAØ%�t—~‘~ dÑ:¨kÑ:ˆHØ�K‰K˜Ô!à˜SÐØÐÜ/°Ó7ˆFØ<BÖ!C°&¤# f¥+Ð!CÐÐ!CàÐÜÑDÐ3CÔDÔDÌSñ QØ0@ôQô Nô !Øwóð ô Ð#Ó$ qÒ(Ð-?ÐCSÒ-SØ�>Ü$Ð%gÓhÐhà"$�KÜÐ-Ó.´#Ð6FÓ2GÒGð 'C˜Ü$ØCÐDTÐCUð V@Ø@RÐ?SÐSVÐWbÐVcðeóð ð
 ÐØ1˜T×1Ñ1°&ÑJ¸MÑJˆNØ&×*Ñ*¨;Ó7ˆIØ�K‰K˜Ô'ð Ð $Ð"2à`hÐitÑ`uö*ØS\Ô.¨y¸$×:MÑ:MÕNð*Ð&ð *ô $PØ*Ø#Ø"×2Ñ2×BÑBÜÑQ¸8ÀKÑ;PÔQÓQô	$Ð ð ,@ˆDÐ'Ñ(à&×*Ñ*Ð+;¸TÓBˆÜ$¨$¸NÔKˆàÐùòg  IùÚoùò "Dùò8*s   Â"JÃ(JÄ7JÈJc                 ó:   —  | j                   j                  |i |¤ŽS )zÇ
        This method forwards all its arguments to PreTrainedTokenizerFast's [`~PreTrainedTokenizer.batch_decode`]. Please
        refer to the docstring of this method for more information.
        ©rY   Úbatch_decode©rd   Úargsrh   s      r   r›   zMllamaProcessor.batch_decodeS  s    € ð
 +ˆt�~‰~×*Ñ*¨DÐ;°FÑ;Ð;r   c                 ó:   —  | j                   j                  |i |¤ŽS )zÁ
        This method forwards all its arguments to PreTrainedTokenizerFast's [`~PreTrainedTokenizer.decode`]. Please refer to
        the docstring of this method for more information.
        )rY   Údecoderœ   s      r   rŸ   zMllamaProcessor.decodeZ  s    € ð
 %ˆt�~‰~×$Ñ$ dÐ5¨fÑ5Ð5r   c                 óB   —  | j                   j                  |f||dœ|¤ŽS )aš  
        Post-process the output of the model to decode the text.

        Args:
            generated_outputs (`torch.Tensor` or `np.ndarray`):
                The output of the model `generate` function. The output is expected to be a tensor of shape `(batch_size, sequence_length)`
                or `(sequence_length,)`.
            skip_special_tokens (`bool`, *optional*, defaults to `True`):
                Whether or not to remove special tokens in the output. Argument passed to the tokenizer's `batch_decode` method.
            Clean_up_tokenization_spaces (`bool`, *optional*, defaults to `False`):
                Whether or not to clean up the tokenization spaces. Argument passed to the tokenizer's `batch_decode` method.
            **kwargs:
                Additional arguments to be passed to the tokenizer's `batch_decode method`.

        Returns:
            `List[str]`: The decoded text.
        )Úskip_special_tokensÚclean_up_tokenization_spacesrš   )rd   Úgenerated_outputsr¡   r¢   rh   s        r   Úpost_process_image_text_to_textz/MllamaProcessor.post_process_image_text_to_texta  s5   € ð( +ˆt�~‰~×*Ñ*Øð
à 3Ø)Eñ
ð ñ	
ð 	
r   c                 ó²   — | j                   j                  }| j                  j                  }|D �cg c]
  }|dk7  sŒ	|‘Œ }}t        ||z   dgz   «      S c c}w )Nr6   rE   )rY   Úmodel_input_namesrX   rƒ   )rd   Útokenizer_input_namesÚimage_processor_input_namesÚnames       r   r¦   z!MllamaProcessor.model_input_names|  se   € à $§¡× @Ñ @ÐØ&*×&:Ñ&:×&LÑ&LÐ#ð 9TÖ&k°ÐW[Ð_jÓWj¢tÐ&kÐ#Ð&kÜÐ)Ð,GÑGÐKaÐJbÑbÓcÐcùò 'ls
   ±
A¼Arn   )NNNN)TF)r   r   r   Ú__doc__Ú
attributesÚvalid_kwargsÚimage_processor_classÚtokenizer_classrc   r   r   r   r   r   r   r   r   r   r˜   r›   rŸ   r¤   Úpropertyr¦   Ú__classcell__)re   s   @r   rW   rW   ²   sÁ   ø„ ñð> $ [Ð1€JØ#Ð$€LØ2ÐØ/€OõRð (,ØhlØØñmà˜Ñ$ðmð �u˜YÐ(9¸4À	¹?ÈDÐQbÑLcÐcÑdÑeðmð Ð.Ñ/ðmð 
ómò^<ò6ð Y^ó
ð6 ñdó ôdr   rW   )rª   Útypingr   r   r   Únumpyr>   Úfeature_extraction_utilsr   Úimage_utilsr   r	   Úprocessing_utilsr
   r   r   r   Útokenization_utils_baser   r   r   r   r   r4   ÚndarrayrN   rp   rU   rW   Ú__all__r   r   r   ú<module>r¹      sü   ðñ  "ç (Ñ (ã å 4ß Aß VÓ V÷ô#˜¨Uõ #ôÐ,°Eõ ð-¨d°3©ið -Èð -ÐQUÐVZÐ[^ÑV_ÑQ`ó -ð`- Ø $ T¨$¨s©)¡_Ñ 5ð- à�D˜‘I‰ð- ð ð- ð ð	- ð
 ‡Z�Zó- ð`"K Cð "K°Cð "KÀcð "KÈcó "KôJRd�nô Rdðj Ð
�r   