Ë
    T^(hïG  ã                   ób  — d Z ddlmZmZmZmZ ddlZddlm	Z	m
Z
mZ ddlmZmZmZ ddlmZmZmZmZmZmZmZmZmZmZmZ ddlmZmZmZm Z m!Z!m"Z"  e «       rddl#Z# e«       rddl$Z$ e!jJ                  e&«      Z'd	„ Z(	 dd
ejR                  dee*   dee*   deeee*f      fd„Z+ G d„ de	«      Z,dgZ-y)z%Image processor class for LayoutLMv3.é    )ÚDictÚIterableÚOptionalÚUnionNé   )ÚBaseImageProcessorÚBatchFeatureÚget_size_dict)ÚresizeÚto_channel_dimension_formatÚto_pil_image)ÚIMAGENET_STANDARD_MEANÚIMAGENET_STANDARD_STDÚChannelDimensionÚ
ImageInputÚPILImageResamplingÚinfer_channel_dimension_formatÚis_scaled_imageÚmake_list_of_imagesÚto_numpy_arrayÚvalid_imagesÚvalidate_preprocess_arguments)Ú
TensorTypeÚfilter_out_non_signature_kwargsÚis_pytesseract_availableÚis_vision_availableÚloggingÚrequires_backendsc                 óž   — t        d| d   |z  z  «      t        d| d   |z  z  «      t        d| d   |z  z  «      t        d| d   |z  z  «      gS )Niè  r   é   é   r   )Úint)ÚboxÚwidthÚheights      úx/var/www/skyplay_api_hub/venv/lib/python3.12/site-packages/transformers/models/layoutlmv3/image_processing_layoutlmv3.pyÚnormalize_boxr'   8   s`   € äˆD�C˜‘F˜U‘NÑ#Ó$ÜˆD�C˜‘F˜V‘OÑ$Ó%ÜˆD�C˜‘F˜U‘NÑ#Ó$ÜˆD�C˜‘F˜V‘OÑ$Ó%ð	ð ó    ÚimageÚlangÚtesseract_configÚinput_data_formatc                 ó˜  — t        | |¬«      }|j                  \  }}t        j                  ||d|¬«      }|d   |d   |d   |d   |d   f\  }}	}
}}t	        |«      D ��cg c]  \  }}|j                  «       rŒ|‘Œ }}}t	        |«      D ��cg c]  \  }}||vsŒ|‘Œ }}}t	        |	«      D ��cg c]  \  }}||vsŒ|‘Œ }	}}t	        |
«      D ��cg c]  \  }}||vsŒ|‘Œ }
}}t	        |«      D ��cg c]  \  }}||vsŒ|‘Œ }}}t	        |«      D ��cg c]  \  }}||vsŒ|‘Œ }}}g }t        |	|
||«      D ]$  \  }}}}||||z   ||z   g}|j                  |«       Œ& g }|D ]  }|j                  t        |||«      «       Œ  t        |«      t        |«      k(  sJ d	«       ‚||fS c c}}w c c}}w c c}}w c c}}w c c}}w c c}}w )
zdApplies Tesseract OCR on a document image, and returns recognized words + normalized bounding boxes.©r,   Údict)r*   Úoutput_typeÚconfigÚtextÚleftÚtopr$   r%   z-Not as many words as there are bounding boxes)
r   ÚsizeÚpytesseractÚimage_to_dataÚ	enumerateÚstripÚzipÚappendr'   Úlen)r)   r*   r+   r,   Ú	pil_imageÚimage_widthÚimage_heightÚdataÚwordsr3   r4   r$   r%   ÚidxÚwordÚirrelevant_indicesÚcoordÚactual_boxesÚxÚyÚwÚhÚ
actual_boxÚnormalized_boxesr#   s                            r&   Úapply_tesseractrM   A   s   € ô ˜UÐ6GÔH€IØ )§¡Ñ€K�Ü×$Ñ$ Y°TÀvÐVfÔg€DØ&*¨6¡l°D¸±LÀ$ÀuÁ+ÈtÐT[É}Ð^bÐckÑ^lÐ&lÑ#€Eˆ4��e˜Vô 09¸Ó/?×T¡) # tÀtÇzÁzÅ|š#ÐTÐÑTÜ#,¨UÓ#3×U‘i�c˜4°sÐBTÒ7TŠTÐU€EÑUÜ$-¨d£O×U‘j�c˜5°sÐBTÒ7TŠEÐU€DÑUÜ#,¨S£>×
S‘Z�S˜%°SÐ@RÒ5RŠ5Ð
S€CÑ
SÜ%.¨uÓ%5×W‘z�s˜E¸ÐDVÒ9VŠUÐW€EÑWÜ&/°Ó&7×Y™
˜˜U¸3ÐFXÒ;XŠeÐY€FÑYð €LÜ˜$  U¨FÓ3ò (‰
ˆˆ1ˆa�Ø˜˜A ™E 1 q¡5Ð)ˆ
Ø×Ñ˜JÕ'ð(ð
 ÐØò OˆØ×Ñ¤¨c°;ÀÓ MÕNðOô ˆu‹:œÐ-Ó.Ò.Ð_Ð0_Ó_Ð.àÐ"Ð"Ð"ùó) UùÛUùÛUùÛ
SùÛWùÛYsH   Á F(Á9F(ÂF.ÂF.Â0F4Â=F4ÃF:ÃF:Ã4G ÄG ÄGÄ#Gc            !       óp  ‡ — e Zd ZdZdgZddej                  ddddddddfdedee	e
f   d	ed
edededeeee   f   deeee   f   dedee	   dee	   ddfˆ fd„Zej                  ddfdej"                  dee	e
f   d	edeee	ef      deee	ef      dej"                  fd„Z e«       ddddddddddddej*                  dfdedee   dee	e
f   d
ee   dee   dee   deeee   f   deeee   f   dee   dee	   dee	   deee	ef      dedeee	ef      dej2                  j2                  fd„«       Zˆ xZS )ÚLayoutLMv3ImageProcessora­
  
    Constructs a LayoutLMv3 image processor.

    Args:
        do_resize (`bool`, *optional*, defaults to `True`):
            Whether to resize the image's (height, width) dimensions to `(size["height"], size["width"])`. Can be
            overridden by `do_resize` in `preprocess`.
        size (`Dict[str, int]` *optional*, defaults to `{"height": 224, "width": 224}`):
            Size of the image after resizing. Can be overridden by `size` in `preprocess`.
        resample (`PILImageResampling`, *optional*, defaults to `PILImageResampling.BILINEAR`):
            Resampling filter to use if resizing the image. Can be overridden by `resample` in `preprocess`.
        do_rescale (`bool`, *optional*, defaults to `True`):
            Whether to rescale the image's pixel values by the specified `rescale_value`. Can be overridden by
            `do_rescale` in `preprocess`.
        rescale_factor (`float`, *optional*, defaults to 1 / 255):
            Value by which the image's pixel values are rescaled. Can be overridden by `rescale_factor` in
            `preprocess`.
        do_normalize (`bool`, *optional*, defaults to `True`):
            Whether to normalize the image. Can be overridden by the `do_normalize` parameter in the `preprocess`
            method.
        image_mean (`Iterable[float]` or `float`, *optional*, defaults to `IMAGENET_STANDARD_MEAN`):
            Mean to use if normalizing the image. This is a float or list of floats the length of the number of
            channels in the image. Can be overridden by the `image_mean` parameter in the `preprocess` method.
        image_std (`Iterable[float]` or `float`, *optional*, defaults to `IMAGENET_STANDARD_STD`):
            Standard deviation to use if normalizing the image. This is a float or list of floats the length of the
            number of channels in the image. Can be overridden by the `image_std` parameter in the `preprocess` method.
        apply_ocr (`bool`, *optional*, defaults to `True`):
            Whether to apply the Tesseract OCR engine to get words + normalized bounding boxes. Can be overridden by
            the `apply_ocr` parameter in the `preprocess` method.
        ocr_lang (`str`, *optional*):
            The language, specified by its ISO code, to be used by the Tesseract OCR engine. By default, English is
            used. Can be overridden by the `ocr_lang` parameter in the `preprocess` method.
        tesseract_config (`str`, *optional*):
            Any additional custom configuration flags that are forwarded to the `config` parameter when calling
            Tesseract. For example: '--psm 6'. Can be overridden by the `tesseract_config` parameter in the
            `preprocess` method.
    Úpixel_valuesTNgp?Ú Ú	do_resizer5   ÚresampleÚ
do_rescaleÚrescale_valueÚdo_normalizeÚ
image_meanÚ	image_stdÚ	apply_ocrÚocr_langr+   Úreturnc                 ó  •— t        ‰| �  di |¤Ž |�|ndddœ}t        |«      }|| _        || _        || _        || _        || _        || _        |�|nt        | _
        |�|nt        | _        |	| _        |
| _        || _        y )Néà   )r%   r$   © )ÚsuperÚ__init__r
   rR   r5   rS   rT   Úrescale_factorrV   r   rW   r   rX   rY   rZ   r+   )ÚselfrR   r5   rS   rT   rU   rV   rW   rX   rY   rZ   r+   ÚkwargsÚ	__class__s                €r&   r`   z!LayoutLMv3ImageProcessor.__init__�   s�   ø€ ô 	‰ÑÑ"˜6Ò"ØÐ'‰t¸ÀcÑ-JˆÜ˜TÓ"ˆà"ˆŒØˆŒ	Ø ˆŒØ$ˆŒØ+ˆÔØ(ˆÔØ(2Ð(>™*ÔDZˆŒØ&/Ð&;™ÔAVˆŒØ"ˆŒØ ˆŒØ 0ˆÕr(   r)   Údata_formatr,   c                 ó–   — t        |«      }d|vsd|vrt        d|j                  «       › �«      ‚|d   |d   f}t        |f||||dœ|¤ŽS )a�  
        Resize an image to `(size["height"], size["width"])`.

        Args:
            image (`np.ndarray`):
                Image to resize.
            size (`Dict[str, int]`):
                Dictionary in the format `{"height": int, "width": int}` specifying the size of the output image.
            resample (`PILImageResampling`, *optional*, defaults to `PILImageResampling.BILINEAR`):
                `PILImageResampling` filter to use when resizing the image e.g. `PILImageResampling.BILINEAR`.
            data_format (`ChannelDimension` or `str`, *optional*):
                The channel dimension format for the output image. If unset, the channel dimension format of the input
                image is used. Can be one of:
                - `"channels_first"` or `ChannelDimension.FIRST`: image in (num_channels, height, width) format.
                - `"channels_last"` or `ChannelDimension.LAST`: image in (height, width, num_channels) format.
                - `"none"` or `ChannelDimension.NONE`: image in (height, width) format.
            input_data_format (`ChannelDimension` or `str`, *optional*):
                The channel dimension format for the input image. If unset, the channel dimension format is inferred
                from the input image. Can be one of:
                - `"channels_first"` or `ChannelDimension.FIRST`: image in (num_channels, height, width) format.
                - `"channels_last"` or `ChannelDimension.LAST`: image in (height, width, num_channels) format.
                - `"none"` or `ChannelDimension.NONE`: image in (height, width) format.

        Returns:
            `np.ndarray`: The resized image.
        r%   r$   zFThe `size` dictionary must contain the keys `height` and `width`. Got )r5   rS   re   r,   )r
   Ú
ValueErrorÚkeysr   )rb   r)   r5   rS   re   r,   rc   Úoutput_sizes           r&   r   zLayoutLMv3ImageProcessor.resize°   sy   € ôF ˜TÓ"ˆØ˜4Ñ 7°$Ñ#6ÜÐeÐfj×foÑfoÓfqÐerÐsÓtÐtØ˜H‘~ t¨G¡}Ð5ˆÜØð
àØØ#Ø/ñ
ð ñ
ð 	
r(   Úimagesra   Úreturn_tensorsc           
      óŒ  — |�|n| j                   }|�|n| j                  }t        |«      }|�|n| j                  }|�|n| j                  }|�|n| j
                  }|�|n| j                  }|�|n| j                  }|	�|	n| j                  }	|
�|
n| j                  }
|�|n| j                  }|�|n| j                  }t        |«      }t        |«      st        d«      ‚t        |||||	|||¬«       |D �cg c]  }t!        |«      ‘Œ }}|r#t#        |d   «      rt$        j'                  d«       |€t)        |d   «      }|
rKt+        | d«       g }g }|D ]6  }t-        ||||¬«      \  }}|j/                  |«       |j/                  |«       Œ8 |r"|D �cg c]  }| j1                  ||||¬«      ‘Œ }}|r!|D �cg c]  }| j3                  |||¬«      ‘Œ }}|r"|D �cg c]  }| j5                  |||	|¬	«      ‘Œ }}|D �cg c]  }t7        |||¬
«      ‘Œ }}t9        d|i|¬«      }|
r
|d<   |d<   |S c c}w c c}w c c}w c c}w c c}w )a%  
        Preprocess an image or batch of images.

        Args:
            images (`ImageInput`):
                Image to preprocess. Expects a single or batch of images with pixel values ranging from 0 to 255. If
                passing in images with pixel values between 0 and 1, set `do_rescale=False`.
            do_resize (`bool`, *optional*, defaults to `self.do_resize`):
                Whether to resize the image.
            size (`Dict[str, int]`, *optional*, defaults to `self.size`):
                Desired size of the output image after applying `resize`.
            resample (`int`, *optional*, defaults to `self.resample`):
                Resampling filter to use if resizing the image. This can be one of the `PILImageResampling` filters.
                Only has an effect if `do_resize` is set to `True`.
            do_rescale (`bool`, *optional*, defaults to `self.do_rescale`):
                Whether to rescale the image pixel values between [0, 1].
            rescale_factor (`float`, *optional*, defaults to `self.rescale_factor`):
                Rescale factor to apply to the image pixel values. Only has an effect if `do_rescale` is set to `True`.
            do_normalize (`bool`, *optional*, defaults to `self.do_normalize`):
                Whether to normalize the image.
            image_mean (`float` or `Iterable[float]`, *optional*, defaults to `self.image_mean`):
                Mean values to be used for normalization. Only has an effect if `do_normalize` is set to `True`.
            image_std (`float` or `Iterable[float]`, *optional*, defaults to `self.image_std`):
                Standard deviation values to be used for normalization. Only has an effect if `do_normalize` is set to
                `True`.
            apply_ocr (`bool`, *optional*, defaults to `self.apply_ocr`):
                Whether to apply the Tesseract OCR engine to get words + normalized bounding boxes.
            ocr_lang (`str`, *optional*, defaults to `self.ocr_lang`):
                The language, specified by its ISO code, to be used by the Tesseract OCR engine. By default, English is
                used.
            tesseract_config (`str`, *optional*, defaults to `self.tesseract_config`):
                Any additional custom configuration flags that are forwarded to the `config` parameter when calling
                Tesseract.
            return_tensors (`str` or `TensorType`, *optional*):
                The type of tensors to return. Can be one of:
                    - Unset: Return a list of `np.ndarray`.
                    - `TensorType.TENSORFLOW` or `'tf'`: Return a batch of type `tf.Tensor`.
                    - `TensorType.PYTORCH` or `'pt'`: Return a batch of type `torch.Tensor`.
                    - `TensorType.NUMPY` or `'np'`: Return a batch of type `np.ndarray`.
                    - `TensorType.JAX` or `'jax'`: Return a batch of type `jax.numpy.ndarray`.
            data_format (`ChannelDimension` or `str`, *optional*, defaults to `ChannelDimension.FIRST`):
                The channel dimension format for the output image. Can be one of:
                    - `ChannelDimension.FIRST`: image in (num_channels, height, width) format.
                    - `ChannelDimension.LAST`: image in (height, width, num_channels) format.
            input_data_format (`ChannelDimension` or `str`, *optional*):
                The channel dimension format for the input image. If unset, the channel dimension format is inferred
                from the input image. Can be one of:
                - `"channels_first"` or `ChannelDimension.FIRST`: image in (num_channels, height, width) format.
                - `"channels_last"` or `ChannelDimension.LAST`: image in (height, width, num_channels) format.
                - `"none"` or `ChannelDimension.NONE`: image in (height, width) format.
        zkInvalid image type. Must be of type PIL.Image.Image, numpy.ndarray, torch.Tensor, tf.Tensor or jax.ndarray.)rT   ra   rV   rW   rX   rR   r5   rS   r   z­It looks like you are trying to rescale already rescaled images. If the input images have pixel values between 0 and 1, set `do_rescale=False` to avoid rescaling them again.r6   r.   )r)   r5   rS   r,   )r)   Úscaler,   )r)   ÚmeanÚstdr,   )Úinput_channel_dimrP   )r@   Útensor_typerA   Úboxes)rR   r5   r
   rS   rT   ra   rV   rW   rX   rY   rZ   r+   r   r   rg   r   r   r   ÚloggerÚwarning_oncer   r   rM   r;   r   ÚrescaleÚ	normalizer   r	   )rb   rj   rR   r5   rS   rT   ra   rV   rW   rX   rY   rZ   r+   rk   re   r,   r)   Úwords_batchÚboxes_batchrA   rr   r@   s                         r&   Ú
preprocessz#LayoutLMv3ImageProcessor.preprocessà   sË  € ðL "+Ð!6‘I¸D¿N¹Nˆ	ØÐ'‰t¨T¯Y©YˆÜ˜TÓ"ˆØ'Ð3‘8¸¿¹ˆØ#-Ð#9‘Z¸t¿¹ˆ
Ø+9Ð+E™È4×K^ÑK^ˆØ'3Ð'?‘|ÀT×EVÑEVˆØ#-Ð#9‘Z¸t¿¹ˆ
Ø!*Ð!6‘I¸D¿N¹Nˆ	Ø!*Ð!6‘I¸D¿N¹Nˆ	Ø'Ð3‘8¸¿¹ˆØ/?Ð/KÑ+ÐQU×QfÑQfÐÜ$ VÓ,ˆä˜FÔ#Üð:óð ô 	&Ø!Ø)Ø%Ø!ØØØØõ		
ð 6<Ö<¨E”. Õ'Ð<ˆÐ<áœ/¨&°©)Ô4Ü×Ñðsôð
 Ð$ä >¸vÀa¹yÓ IÐñ Ü˜d MÔ2ØˆKØˆKØò *�Ü.¨u°hÐ@PÐduÔv‘��uØ×"Ñ" 5Ô)Ø×"Ñ" 5Õ)ð*ñ
 ð $öàð —‘ %¨d¸XÐYj�ÕkðˆFð ñ
 ð $öàð —‘ 5°ÐRc�ÕdðˆFð ñ
 ð $öàð —‘ U°ÀÐ^o�ÕpðˆFð ð ouö
ØejÔ'¨¨{ÐN_Ö`ð
ˆð 
ô  .°&Ð!9À~ÔVˆáØ'ˆD�‰MØ'ˆD�‰MØˆùòc =ùò.ùòùòùò

s   Ã4H-ÆH2Æ4H7ÇH<Ç9I)Ú__name__Ú
__module__Ú__qualname__Ú__doc__Úmodel_input_namesr   ÚBILINEARÚboolr   Ústrr"   Úfloatr   r   r   r`   ÚnpÚndarrayr   r   r   ÚFIRSTr   r   ÚPILÚImagery   Ú__classcell__)rd   s   @r&   rO   rO   g   sº  ø„ ñ$ðL (Ð(Ðð Ø#Ø'9×'BÑ'BØØ&Ø!Ø48Ø37ØØ"&Ø*,ñ1àð1ð �3˜�8‰nð1ð %ð	1ð
 ð1ð ð1ð ð1ð ˜% ¨%¡Ð0Ñ1ð1ð ˜ ¨¡Ð/Ñ0ð1ð ð1ð ˜3‘-ð1ð # 3™-ð1ð 
õ1ðH (:×'BÑ'BØ>BØDHñ.
à�z‰zð.
ð �3˜�8‰nð.
ð %ð	.
ð
 ˜e CÐ)9Ð$9Ñ:Ñ;ð.
ð $ E¨#Ð/?Ð*?Ñ$@ÑAð.
ð 
�‰ó.
ñ` %Ó&ð %)Ø#ØØ%)Ø*.Ø'+Ø48Ø37Ø$(Ø"&Ø*.Ø;?Ø(8×(>Ñ(>ØDHñ!UàðUð ˜D‘>ðUð �3˜�8‰nð	Uð ˜T‘NðUð ! ™ðUð ˜t‘nðUð ˜% ¨%¡Ð0Ñ1ðUð ˜ ¨¡Ð/Ñ0ðUð ˜D‘>ðUð ˜3‘-ðUð # 3™-ðUð !  s¨J Ñ!7Ñ8ðUð &ðUð  $ E¨#Ð/?Ð*?Ñ$@ÑAð!Uð" 
�‰�‰ò#Uó 'ôUr(   rO   )N).r}   Útypingr   r   r   r   Únumpyrƒ   Úimage_processing_utilsr   r	   r
   Úimage_transformsr   r   r   Úimage_utilsr   r   r   r   r   r   r   r   r   r   r   Úutilsr   r   r   r   r   r   r†   r6   Ú
get_loggerrz   rs   r'   r„   r�   rM   rO   Ú__all__r^   r(   r&   ú<module>r‘      sÑ   ðñ ,ç 2Ó 2ã ç UÑ Uß QÑ Q÷÷ ÷ ñ ÷÷ ñ ÔÛñ ÔÛà	ˆ×	Ñ	˜HÓ	%€òð AEñ	##Ø�:‰:ð##à
�3‰-ð##ð ˜s‘mð##ð   Ð&6¸Ð&;Ñ <Ñ=ó	##ôLOÐ1ô Oðd &Ð
&�r(   