Ë
    T^(hCM  ã                   óÞ  — d Z ddlZddlZddlmZmZmZ ddlZddl	m
Z
 ddlmZmZ ddlmZmZmZmZ ddlmZmZmZmZmZmZmZ dd	lmZmZmZmZ dd
l m!Z!  e«       rddl"Z"ddl#m$Z$m%Z%m&Z&  e«       rddl'Z' ejP                  e)«      Z*dZ+d„ Z,	 	 	 	 	 	 	 	 	 d de-de.de-de-de.de.de.de.dee/   dee-   de$jH                  fd„Z0	 d!dejb                  de-deee-e2f      fd„Z3 G d„ de«      Z4dgZ5y)"z%Image processor class for Pix2Struct.é    N)ÚDictÚOptionalÚUnion)Úhf_hub_downloadé   )ÚBaseImageProcessorÚBatchFeature)Úconvert_to_rgbÚ	normalizeÚto_channel_dimension_formatÚto_pil_image)ÚChannelDimensionÚ
ImageInputÚget_image_sizeÚinfer_channel_dimension_formatÚmake_list_of_imagesÚto_numpy_arrayÚvalid_images)Ú
TensorTypeÚis_torch_availableÚis_vision_availableÚlogging)Úrequires_backends)ÚImageÚ	ImageDrawÚ	ImageFontzybelkada/fontsc                 óì  — t        t        dg«       | j                  d«      } t        j                  j
                  j                  | ||f||f¬«      }|j                  | j                  d«      | j                  d«      ||d«      }|j                  ddddd«      j                  | j                  d«      |z  | j                  d«      |z  | j                  d«      |z  |z  «      }|j                  d«      S )	a¸  
    Utiliy function to extract patches from a given image tensor. Returns a tensor of shape (1, `patch_height`,
    `patch_width`, `num_channels`x `patch_height` x `patch_width`)

    Args:
        image_tensor (torch.Tensor):
            The image tensor to extract patches from.
        patch_height (int):
            The height of the patches to extract.
        patch_width (int):
            The width of the patches to extract.
    Útorchr   )Ústrideé   éÿÿÿÿé   é   r   )
r   Útorch_extract_patchesÚ	unsqueezer   ÚnnÚ
functionalÚunfoldÚreshapeÚsizeÚpermute)Úimage_tensorÚpatch_heightÚpatch_widthÚpatchess       úx/var/www/skyplay_api_hub/venv/lib/python3.12/site-packages/transformers/models/pix2struct/image_processing_pix2struct.pyr$   r$   4   sî   € ô Ô+¨g¨YÔ7à×)Ñ)¨!Ó,€LÜ�h‰h×!Ñ!×(Ñ(¨¸ÀkÐ7RÐ\hÐjuÐ[vÐ(Ów€GØ�o‰o˜l×/Ñ/°Ó2°L×4EÑ4EÀaÓ4HÈ,ÐXcÐegÓh€GØ�o‰o˜a  A q¨!Ó,×4Ñ4Ø×Ñ˜!Ó Ñ,Ø×Ñ˜!Ó Ñ+Ø×Ñ˜!Ó˜|Ñ+¨kÑ9ó€Gð
 ×Ñ˜QÓÐó    ÚtextÚ	text_sizeÚ
text_colorÚbackground_colorÚleft_paddingÚright_paddingÚtop_paddingÚbottom_paddingÚ
font_bytesÚ	font_pathÚreturnc
                 óT  — t        t        d«       t        j                  d¬«      }
|
j	                  | ¬«      }dj                  |«      }|�|	€t        j                  |«      }n|	�|	}nt        t        d«      }t        j                  |d|¬«      }t        j                  t        j                  d	d
|«      «      }|j!                  d||«      \  }}}}||z   |z   }||z   |z   }t        j                  d	||f|«      }t        j                  |«      }|j#                  ||f|||¬«       |S )a£  
    Render text. This script is entirely adapted from the original script that can be found here:
    https://github.com/google-research/pix2struct/blob/main/pix2struct/preprocessing/preprocessing_utils.py

    Args:
        text (`str`, *optional*, defaults to ):
            Text to render.
        text_size (`int`, *optional*, defaults to 36):
            Size of the text.
        text_color (`str`, *optional*, defaults to `"black"`):
            Color of the text.
        background_color (`str`, *optional*, defaults to `"white"`):
            Color of the background.
        left_padding (`int`, *optional*, defaults to 5):
            Padding on the left.
        right_padding (`int`, *optional*, defaults to 5):
            Padding on the right.
        top_padding (`int`, *optional*, defaults to 5):
            Padding on the top.
        bottom_padding (`int`, *optional*, defaults to 5):
            Padding on the bottom.
        font_bytes (`bytes`, *optional*):
            Bytes of the font to use. If `None`, the default font will be used.
        font_path (`str`, *optional*):
            Path to the font to use. If `None`, the default font will be used.
    ÚvisionéP   )Úwidth)r2   ú
z	Arial.TTFzUTF-8)Úencodingr*   ÚRGB)r    r    ©r   r   )Úxyr2   ÚfillÚfont)r   Úrender_textÚtextwrapÚTextWrapperÚwrapÚjoinÚioÚBytesIOr   ÚDEFAULT_FONT_PATHr   Útruetyper   ÚDrawr   ÚnewÚtextbboxr2   )r2   r3   r4   r5   r6   r7   r8   r9   r:   r;   ÚwrapperÚlinesÚwrapped_textrG   Ú	temp_drawÚ_Ú
text_widthÚtext_heightÚimage_widthÚimage_heightÚimageÚdraws                         r0   rH   rH   O   s$  € ôL ”k 8Ô,ô ×"Ñ"¨Ô,€GØ�L‰L˜dˆLÓ#€EØ—9‘9˜UÓ#€LàÐ )Ð"3Ü�z‰z˜*Ó%‰Ø	Ð	Ø‰äÔ0°+Ó>ˆÜ×Ñ˜d¨W¸9ÔE€Dô —‘œuŸy™y¨°Ð8HÓIÓJ€IØ$-×$6Ñ$6°v¸|ÈTÓ$RÑ!€A€qˆ*�kð ˜|Ñ+¨mÑ;€KØ Ñ,¨~Ñ=€LÜ�I‰I�e˜k¨<Ð8Ð:JÓK€EÜ�>‰>˜%Ó €DØ‡I�I�, Ð,°<ÀjÐW[€IÔ\Ø€Lr1   r]   ÚheaderÚinput_data_formatc                 óv  — t        t        d«       t        | |¬«      } t        |fi |¤Ž}t	        |j
                  | j
                  «      }t        | j                  || j
                  z  z  «      }t        |j                  ||j
                  z  z  «      }t        j                  d|||z   fd«      }|j                  |j                  ||f«      d«       |j                  | j                  ||f«      d|f«       t        |«      }t        |«      t        j                  k(  rt!        |t        j                  «      }|S )aâ  
    Renders the input text as a header on the input image.

    Args:
        image (`np.ndarray`):
            The image to render the header on.
        header (`str`):
            The header text.
        data_format (`Union[ChannelDimension, str]`, *optional*):
            The data format of the image. Can be either "ChannelDimension.channels_first" or
            "ChannelDimension.channels_last".

    Returns:
        `np.ndarray`: The image with the header rendered.
    r>   )r`   rC   ÚwhiterD   r   )r   Úrender_headerr   rH   Úmaxr@   ÚintÚheightr   rR   ÚpasteÚresizer   r   r   ÚLASTr   )	r]   r_   r`   ÚkwargsÚheader_imageÚ	new_widthÚ
new_heightÚnew_header_heightÚ	new_images	            r0   rc   rc   “   s  € ô$ ”m XÔ.ô ˜Ð2CÔD€Eä˜vÑ0¨Ñ0€LÜ�L×&Ñ&¨¯©Ó4€Iä�U—\‘\ Y°·±Ñ%<Ñ=Ó>€JÜ˜L×/Ñ/°9¸|×?QÑ?QÑ3QÑRÓSÐä—	‘	˜% )¨ZÐ:KÑ-KÐ!LÈgÓV€IØ‡O�O�L×'Ñ'¨Ð4EÐ(FÓGÈÔPØ‡O�O�E—L‘L )¨ZÐ!8Ó9¸AÐ?PÐ;QÔRô ˜yÓ)€Iä% iÓ0Ô4D×4IÑ4IÒIÜ/°	Ô;K×;PÑ;PÓQˆ	àÐr1   c                   ó´  ‡ — e Zd ZdZdgZ	 	 	 	 	 ddededeeef   deded	dfˆ fd
„Z		 dde
j                  dededeeeef      d	e
j                  f
d„Z	 	 dde
j                  deeeef      deeeef      d	e
j                  fd„Zddddddej$                  dfdedee   dee   dee   dee   deeeef      deeeef      dedeeeef      d	efd„Zˆ xZS )ÚPix2StructImageProcessoraf  
    Constructs a Pix2Struct image processor.

    Args:
        do_convert_rgb (`bool`, *optional*, defaults to `True`):
            Whether to convert the image to RGB.
        do_normalize (`bool`, *optional*, defaults to `True`):
            Whether to normalize the image. Can be overridden by the `do_normalize` parameter in the `preprocess`
            method. According to Pix2Struct paper and code, the image is normalized with its own mean and standard
            deviation.
        patch_size (`Dict[str, int]`, *optional*, defaults to `{"height": 16, "width": 16}`):
            The patch size to use for the image. According to Pix2Struct paper and code, the patch size is 16x16.
        max_patches (`int`, *optional*, defaults to 2048):
            The maximum number of patches to extract from the image as per the [Pix2Struct
            paper](https://arxiv.org/pdf/2210.03347.pdf).
        is_vqa (`bool`, *optional*, defaults to `False`):
            Whether or not the image processor is for the VQA task. If `True` and `header_text` is passed in, text is
            rendered onto the input images.
    Úflattened_patchesNÚdo_convert_rgbÚdo_normalizeÚ
patch_sizeÚmax_patchesÚis_vqar<   c                 óx   •— t        ‰| �  di |¤Ž |�|ndddœ| _        || _        || _        || _        || _        y )Né   )rf   r@   © )ÚsuperÚ__init__ru   rt   rs   rv   rw   )Úselfrs   rt   ru   rv   rw   rj   Ú	__class__s          €r0   r|   z!Pix2StructImageProcessor.__init__Ô   sH   ø€ ô 	‰ÑÑ"˜6Ò"Ø(2Ð(>™*ÈrÐ\^ÑD_ˆŒØ(ˆÔØ,ˆÔØ&ˆÔØˆ�r1   r]   r`   c           	      ó¶  — t        | j                  d«       t        |t        j                  |«      }t        j                  |«      }|d   |d   }}t        |t        j                  «      \  }}	t        j                  |||z  z  ||	z  z  «      }
t        t        t        j                  |
|z  |z  «      |«      d«      }t        t        t        j                  |
|	z  |z  «      |«      d«      }t        ||z  d«      }t        ||z  d«      }t
        j                  j                  j                  |j!                  d«      ||fddd¬	«      j#                  d«      }t%        |||«      }|j&                  }|d   }|d
   }|d   }|j)                  ||z  |g«      }t        j*                  |«      j)                  |dg«      j-                  d|«      j)                  ||z  dg«      }t        j*                  |«      j)                  d|g«      j-                  |d«      j)                  ||z  dg«      }|dz  }|dz  }|j/                  t
        j0                  «      }|j/                  t
        j0                  «      }t        j2                  |||gd«      }t
        j                  j                  j5                  |ddd|||z  z
  g«      j7                  «       }t9        |«      }|S )aÒ  
        Extract flattened patches from an image.

        Args:
            image (`np.ndarray`):
                Image to extract flattened patches from.
            max_patches (`int`):
                Maximum number of patches to extract.
            patch_size (`dict`):
                Dictionary containing the patch height and width.

        Returns:
            result (`np.ndarray`):
                A sequence of `max_patches` flattened patches.
        r   rf   r@   r    r   ÚbilinearFT)r*   ÚmodeÚalign_cornersÚ	antialiasr#   r   r!   )r   Úextract_flattened_patchesr   r   ÚFIRSTr   Ú
from_numpyr   ÚmathÚsqrtrd   ÚminÚfloorr&   r'   Úinterpolater%   Úsqueezer$   Úshaper)   ÚarangeÚrepeatÚtoÚfloat32ÚcatÚpadÚfloatr   )r}   r]   rv   ru   r`   rj   r-   r.   r\   r[   ÚscaleÚnum_feasible_rowsÚnum_feasible_colsÚresized_heightÚresized_widthr/   Úpatches_shapeÚrowsÚcolumnsÚdepthÚrow_idsÚcol_idsÚresults                          r0   r„   z2Pix2StructImageProcessor.extract_flattened_patchesä   s¾  € ô. 	˜$×8Ñ8¸'ÔBô ,¨EÔ3C×3IÑ3IÐK\Ó]ˆÜ× Ñ  Ó'ˆà$.¨xÑ$8¸*ÀWÑ:M�kˆÜ$2°5Ô:J×:PÑ:PÓ$QÑ!ˆ�kô —	‘	˜+¨¸Ñ)DÑEÈÐWbÑIbÑcÓdˆÜ¤¤D§J¡J¨u°|Ñ/CÀlÑ/RÓ$SÐU`Ó aÐcdÓeÐÜ¤¤D§J¡J¨u°{Ñ/BÀ[Ñ/PÓ$QÐS^Ó _ÐabÓcÐÜÐ.°Ñ=¸qÓAˆÜÐ-°Ñ;¸QÓ?ˆä—‘×#Ñ#×/Ñ/Ø�O‰O˜AÓØ  -Ð0ØØØð 0ó 
÷ ‰'�!‹*ð 	ô (¨¨|¸[ÓIˆàŸ™ˆØ˜QÑˆØ Ñ"ˆØ˜aÑ ˆð —/‘/ 4¨'¡>°5Ð"9Ó:ˆô —,‘,˜tÓ$×,Ñ,¨d°A¨YÓ7×>Ñ>¸qÀ'ÓJ×RÑRÐTXÐ[bÑTbÐdeÐSfÓgˆÜ—,‘,˜wÓ'×/Ñ/°°G°Ó=×DÑDÀTÈ1ÓM×UÑUÐW[Ð^eÑWeÐghÐViÓjˆð 	�1‰ˆØ�1‰ˆð —*‘*œUŸ]™]Ó+ˆØ—*‘*œUŸ]™]Ó+ˆô —‘˜G W¨gÐ6¸Ó;ˆô —‘×$Ñ$×(Ñ(¨°!°Q¸¸;È$ÐQXÉ.Ñ;YÐ1ZÓ[×aÑaÓcˆä Ó'ˆàˆr1   Údata_formatc           	      ón  — |j                   t        j                  k(  r|j                  t        j                  «      }t        j
                  |«      }t        j                  |«      }t        |dt        j                  t        j                  |j                  «      «      z  «      }t        |f||||dœ|¤ŽS )aè  
        Normalize an image. image = (image - image_mean) / image_std.

        The image std is to mimic the tensorflow implementation of the `per_image_standardization`:
        https://www.tensorflow.org/api_docs/python/tf/image/per_image_standardization

        Args:
            image (`np.ndarray`):
                Image to normalize.
            data_format (`str` or `ChannelDimension`, *optional*):
                The channel dimension format for the output image. If unset, the channel dimension format of the input
                image is used.
            input_data_format (`str` or `ChannelDimension`, *optional*):
                The channel dimension format of the input image. If not provided, it will be inferred.
        g      ð?)ÚmeanÚstdr¡   r`   )ÚdtypeÚnpÚuint8Úastyper‘   r£   r¤   rd   r‡   rˆ   Úprodr�   r   )r}   r]   r¡   r`   rj   r£   r¤   Úadjusted_stddevs           r0   r   z"Pix2StructImageProcessor.normalize5  s”   € ð, �;‰;œ"Ÿ(™(Ò"Ø—L‘L¤§¡Ó,ˆEô �w‰w�u‹~ˆÜ�f‰f�U‹mˆÜ˜c 3¬¯©´2·7±7¸5¿;¹;Ó3GÓ)HÑ#HÓIˆäØð
àØØ#Ø/ñ
ð ñ
ð 	
r1   ÚimagesÚheader_textÚreturn_tensorsc
           
      ó   — |�|n| j                   }|�|n| j                  }|�|n| j                  }|�|n| j                  }| j                  }|
j                  dd«      �t        d«      ‚t        |«      }t        |«      st        d«      ‚|r|D �cg c]  }t        |«      ‘Œ }}|D �cg c]  }t        |«      ‘Œ }}|	€t        |d   «      }	|r}|€t        d«      ‚|
j                  dd«      }|
j                  dd«      }t        |t        «      r|gt        |«      z  }t!        |«      D ��cg c]  \  }}t#        |||   ||¬	«      ‘Œ }}}|r |D �cg c]  }| j%                  ||	¬
«      ‘Œ }}|D �cg c]  }| j'                  ||||	¬«      ‘Œ }}|D �cg c]4  }|j)                  d¬«      dk7  j+                  t,        j.                  «      ‘Œ6 }}t1        ||dœ|¬«      }|S c c}w c c}w c c}}w c c}w c c}w c c}w )a›  
        Preprocess an image or batch of images. The processor first computes the maximum possible number of
        aspect-ratio preserving patches of size `patch_size` that can be extracted from the image. It then pads the
        image with zeros to make the image respect the constraint of `max_patches`. Before extracting the patches the
        images are standardized following the tensorflow implementation of `per_image_standardization`
        (https://www.tensorflow.org/api_docs/python/tf/image/per_image_standardization).


        Args:
            images (`ImageInput`):
                Image to preprocess. Expects a single or batch of images.
            header_text (`Union[List[str], str]`, *optional*):
                Text to render as a header. Only has an effect if `image_processor.is_vqa` is `True`.
            do_convert_rgb (`bool`, *optional*, defaults to `self.do_convert_rgb`):
                Whether to convert the image to RGB.
            do_normalize (`bool`, *optional*, defaults to `self.do_normalize`):
                Whether to normalize the image.
            max_patches (`int`, *optional*, defaults to `self.max_patches`):
                Maximum number of patches to extract.
            patch_size (`dict`, *optional*, defaults to `self.patch_size`):
                Dictionary containing the patch height and width.
            return_tensors (`str` or `TensorType`, *optional*):
                The type of tensors to return. Can be one of:
                    - Unset: Return a list of `np.ndarray`.
                    - `TensorType.TENSORFLOW` or `'tf'`: Return a batch of type `tf.Tensor`.
                    - `TensorType.PYTORCH` or `'pt'`: Return a batch of type `torch.Tensor`.
                    - `TensorType.NUMPY` or `'np'`: Return a batch of type `np.ndarray`.
                    - `TensorType.JAX` or `'jax'`: Return a batch of type `jax.numpy.ndarray`.
            data_format (`ChannelDimension` or `str`, *optional*, defaults to `ChannelDimension.FIRST`):
                The channel dimension format for the output image. Can be one of:
                - `"channels_first"` or `ChannelDimension.FIRST`: image in (num_channels, height, width) format.
                - `"channels_last"` or `ChannelDimension.LAST`: image in (height, width, num_channels) format.
                - Unset: Use the channel dimension format of the input image.
            input_data_format (`ChannelDimension` or `str`, *optional*):
                The channel dimension format for the input image. If unset, the channel dimension format is inferred
                from the input image. Can be one of:
                - `"channels_first"` or `ChannelDimension.FIRST`: image in (num_channels, height, width) format.
                - `"channels_last"` or `ChannelDimension.LAST`: image in (height, width, num_channels) format.
                - `"none"` or `ChannelDimension.NONE`: image in (height, width) format.
        Nr¡   z8data_format is not an accepted input as the outputs are zkInvalid image type. Must be of type PIL.Image.Image, numpy.ndarray, torch.Tensor, tf.Tensor or jax.ndarray.r   z.A header text must be provided for VQA models.r:   r;   )r:   r;   )r]   r`   )r]   rv   ru   r`   r!   )Úaxis)rr   Úattention_mask)ÚdataÚtensor_type)rt   rs   ru   rv   rw   ÚgetÚ
ValueErrorr   r   r
   r   r   ÚpopÚ
isinstanceÚstrÚlenÚ	enumeraterc   r   r„   Úsumr¨   r¦   r‘   r	   )r}   r«   r¬   rs   rt   rv   ru   r­   r¡   r`   rj   rw   r]   r:   r;   ÚiÚattention_masksÚencoded_outputss                     r0   Ú
preprocessz#Pix2StructImageProcessor.preprocess\  s?  € ðj (4Ð'?‘|ÀT×EVÑEVˆØ+9Ð+E™È4×K^ÑK^ˆØ#-Ð#9‘Z¸t¿¹ˆ
Ø%0Ð%<‘kÀ$×BRÑBRˆØ—‘ˆà�:‰:�m TÓ*Ð6ÜÐWÓXÐXä$ VÓ,ˆä˜FÔ#Üð:óð ñ Ø9?Ö@°”n UÕ+Ð@ˆFÐ@ð 6<Ö<¨E”. Õ'Ð<ˆÐ<àÐ$ä >¸vÀa¹yÓ IÐáØÐ"Ü Ð!QÓRÐRØŸ™ L°$Ó7ˆJØŸ
™
 ;°Ó5ˆIä˜+¤sÔ+Ø*˜m¬c°&«kÑ9�ô !*¨&Ó 1÷á�A�uô ˜e [°¡^À
ÐV_Ö`ðˆFñ ñ
 ØdjÖkÐ[`�d—n‘n¨5ÐDU�nÕVÐkˆFÐkð  ö	
ð ð ×*Ñ*Ø¨ÀÐ_pð +õ ð
ˆð 
ð V\Ö\ÈE˜EŸI™I¨2˜IÓ.°!Ñ3×;Ñ;¼B¿J¹JÕGÐ\ˆÐ\ä&Ø'-ÀÑQÐ_mô
ˆð ÐùòS Aùò =ùóùò lùò
ùò ]s$   ÂG!Â)G&Ä-G+ÅG1Å2G6Æ9G;)TTNi   F©N)NN)Ú__name__Ú
__module__Ú__qualname__Ú__doc__Úmodel_input_namesÚboolr   r·   re   r|   r¦   ÚndarrayÚdictr   r   r   r„   r   r…   r   r   r¾   Ú__classcell__)r~   s   @r0   rq   rq   ½   sð  ø„ ñð( -Ð-Ðð  $Ø!Ø%)ØØñàðð ðð ˜˜c˜‘Nð	ð
 ðð ðð 
õð* EIñOà�z‰zðOð ðOð ð	Oð
 $ E¨#Ð/?Ð*?Ñ$@ÑAðOð 
�‰óOðh ?CØDHñ	%
à�z‰zð%
ð ˜e CÐ)9Ð$9Ñ:Ñ;ð%
ð $ E¨#Ð/?Ð*?Ñ$@ÑAð	%
ð 
�‰ó%
ðT &*Ø)-Ø'+Ø%)Ø/3Ø;?Ø(8×(>Ñ(>ØDHñqàðqð ˜c‘]ðqð ! ™ð	qð
 ˜t‘nðqð ˜c‘]ðqð ˜T # s (™^Ñ,ðqð !  s¨J Ñ!7Ñ8ðqð &ðqð $ E¨#Ð/?Ð*?Ñ$@ÑAðqð 
÷qr1   rq   )	é$   Úblackrb   é   rË   rË   rË   NNr¿   )6rÃ   rM   r‡   Útypingr   r   r   Únumpyr¦   Úhuggingface_hubr   Úimage_processing_utilsr   r	   Úimage_transformsr
   r   r   r   Úimage_utilsr   r   r   r   r   r   r   Úutilsr   r   r   r   Úutils.import_utilsr   rI   ÚPILr   r   r   r   Ú
get_loggerrÀ   ÚloggerrO   r$   r·   re   ÚbytesrH   rÆ   ÚChildProcessErrorrc   rq   Ú__all__rz   r1   r0   ú<module>rÚ      so  ðñ ,ã 	Û ß (Ñ (ã Ý +ç Fß dÓ d÷÷ ñ ÷ RÓ QÝ 3ñ ÔÛç/Ñ/áÔÛà	ˆ×	Ñ	˜HÓ	%€Ø$Ð ò ð: ØØ#ØØØØØ"&Ø#ñ@Ø
ð@àð@ð ð@ð ð	@ð
 ð@ð ð@ð ð@ð ð@ð ˜‘ð@ð ˜‰}ð@ð ‡[�[ó@ðJ bfñ'Ø�:‰:ð'Ø"ð'Ø7?ÀÀcÐK\ÐF\Ñ@]Ñ7^ó'ôTPÐ1ô Pðf &Ð
&�r1   