Ë
    T^(hÌ4  ã                   óX  — d Z ddlmZmZmZ ddlZddlmZm	Z	m
Z
 ddlmZmZmZmZ ddlmZmZmZmZmZmZmZmZ ddlmZmZmZmZmZmZ  e«       rddl Z  e«       rddl!Z! ejD                  e#«      Z$d	„ Z%	 	 dd
ejL                  dee'   dee'   deee'ef      fd„Z( G d„ de«      Z)dgZ*y)z%Image processor class for LayoutLMv2.é    )ÚDictÚOptionalÚUnionNé   )ÚBaseImageProcessorÚBatchFeatureÚget_size_dict)Úflip_channel_orderÚresizeÚto_channel_dimension_formatÚto_pil_image)ÚChannelDimensionÚ
ImageInputÚPILImageResamplingÚinfer_channel_dimension_formatÚmake_list_of_imagesÚto_numpy_arrayÚvalid_imagesÚvalidate_preprocess_arguments)Ú
TensorTypeÚfilter_out_non_signature_kwargsÚis_pytesseract_availableÚis_vision_availableÚloggingÚrequires_backendsc                 óž   — t        d| d   |z  z  «      t        d| d   |z  z  «      t        d| d   |z  z  «      t        d| d   |z  z  «      gS )Niè  r   é   é   r   )Úint)ÚboxÚwidthÚheights      úx/var/www/skyplay_api_hub/venv/lib/python3.12/site-packages/transformers/models/layoutlmv2/image_processing_layoutlmv2.pyÚnormalize_boxr$   5   s`   € äˆD�C˜‘F˜U‘NÑ#Ó$ÜˆD�C˜‘F˜V‘OÑ$Ó%ÜˆD�C˜‘F˜U‘NÑ#Ó$ÜˆD�C˜‘F˜V‘OÑ$Ó%ð	ð ó    ÚimageÚlangÚtesseract_configÚinput_data_formatc                 ó¤  — |�|nd}t        | |¬«      }|j                  \  }}t        j                  ||d|¬«      }|d   |d   |d   |d   |d	   f\  }}	}
}}t	        |«      D ��cg c]  \  }}|j                  «       rŒ|‘Œ }}}t	        |«      D ��cg c]  \  }}||vsŒ|‘Œ }}}t	        |	«      D ��cg c]  \  }}||vsŒ|‘Œ }	}}t	        |
«      D ��cg c]  \  }}||vsŒ|‘Œ }
}}t	        |«      D ��cg c]  \  }}||vsŒ|‘Œ }}}t	        |«      D ��cg c]  \  }}||vsŒ|‘Œ }}}g }t        |	|
||«      D ]$  \  }}}}||||z   ||z   g}|j                  |«       Œ& g }|D ]  }|j                  t        |||«      «       Œ  t        |«      t        |«      k(  sJ d
«       ‚||fS c c}}w c c}}w c c}}w c c}}w c c}}w c c}}w )zdApplies Tesseract OCR on a document image, and returns recognized words + normalized bounding boxes.Ú ©r)   Údict)r'   Úoutput_typeÚconfigÚtextÚleftÚtopr!   r"   z-Not as many words as there are bounding boxes)
r   ÚsizeÚpytesseractÚimage_to_dataÚ	enumerateÚstripÚzipÚappendr$   Úlen)r&   r'   r(   r)   Ú	pil_imageÚimage_widthÚimage_heightÚdataÚwordsr1   r2   r!   r"   ÚidxÚwordÚirrelevant_indicesÚcoordÚactual_boxesÚxÚyÚwÚhÚ
actual_boxÚnormalized_boxesr    s                            r#   Úapply_tesseractrK   >   s  € ð ,<Ð+GÑ'ÈRÐô ˜UÐ6GÔH€IØ )§¡Ñ€K�Ü×$Ñ$ Y°TÀvÐVfÔg€DØ&*¨6¡l°D¸±LÀ$ÀuÁ+ÈtÐT[É}Ð^bÐckÑ^lÐ&lÑ#€Eˆ4��e˜Vô 09¸Ó/?×T¡) # tÀtÇzÁzÅ|š#ÐTÐÑTÜ#,¨UÓ#3×U‘i�c˜4°sÐBTÒ7TŠTÐU€EÑUÜ$-¨d£O×U‘j�c˜5°sÐBTÒ7TŠEÐU€DÑUÜ#,¨S£>×
S‘Z�S˜%°SÐ@RÒ5RŠ5Ð
S€CÑ
SÜ%.¨uÓ%5×W‘z�s˜E¸ÐDVÒ9VŠUÐW€EÑWÜ&/°Ó&7×Y™
˜˜U¸3ÐFXÒ;XŠeÐY€FÑYð €LÜ˜$  U¨FÓ3ò (‰
ˆˆ1ˆa�Ø˜˜A ™E 1 q¡5Ð)ˆ
Ø×Ñ˜JÕ'ð(ð
 ÐØò OˆØ×Ñ¤¨c°;ÀÓ MÕNðOô ˆu‹:œÐ-Ó.Ò.Ð_Ð0_Ó_Ð.àÐ"Ð"Ð"ùó) UùÛUùÛUùÛ
SùÛWùÛYsH   Á&F.Á?F.ÂF4Â!F4Â6F:ÃF:ÃG Ã%G Ã:GÄGÄGÄ)Gc                   óæ  ‡ — e Zd ZdZdgZddej                  dddfdedee	e
f   ded	ed
ee	   dee	   ddfˆ fd„Zej                  ddfdej                  dee	e
f   dedeee	ef      deee	ef      dej                  fd„Z e«       dddddddej&                  df	dedee   dee	e
f   ded	ee   d
ee	   dee	   deee	ef      dedeee	ef      dej.                  j.                  fd„«       Zˆ xZS )ÚLayoutLMv2ImageProcessora�  
    Constructs a LayoutLMv2 image processor.

    Args:
        do_resize (`bool`, *optional*, defaults to `True`):
            Whether to resize the image's (height, width) dimensions to `(size["height"], size["width"])`. Can be
            overridden by `do_resize` in `preprocess`.
        size (`Dict[str, int]` *optional*, defaults to `{"height": 224, "width": 224}`):
            Size of the image after resizing. Can be overridden by `size` in `preprocess`.
        resample (`PILImageResampling`, *optional*, defaults to `Resampling.BILINEAR`):
            Resampling filter to use if resizing the image. Can be overridden by the `resample` parameter in the
            `preprocess` method.
        apply_ocr (`bool`, *optional*, defaults to `True`):
            Whether to apply the Tesseract OCR engine to get words + normalized bounding boxes. Can be overridden by
            `apply_ocr` in `preprocess`.
        ocr_lang (`str`, *optional*):
            The language, specified by its ISO code, to be used by the Tesseract OCR engine. By default, English is
            used. Can be overridden by `ocr_lang` in `preprocess`.
        tesseract_config (`str`, *optional*, defaults to `""`):
            Any additional custom configuration flags that are forwarded to the `config` parameter when calling
            Tesseract. For example: '--psm 6'. Can be overridden by `tesseract_config` in `preprocess`.
    Úpixel_valuesTNr+   Ú	do_resizer3   ÚresampleÚ	apply_ocrÚocr_langr(   Úreturnc                 ó    •— t        ‰| �  di |¤Ž |�|ndddœ}t        |«      }|| _        || _        || _        || _        || _        || _        y )Néà   )r"   r!   © )	ÚsuperÚ__init__r	   rO   r3   rP   rQ   rR   r(   )	ÚselfrO   r3   rP   rQ   rR   r(   ÚkwargsÚ	__class__s	           €r#   rX   z!LayoutLMv2ImageProcessor.__init__   s[   ø€ ô 	‰ÑÑ"˜6Ò"ØÐ'‰t¸ÀcÑ-JˆÜ˜TÓ"ˆà"ˆŒØˆŒ	Ø ˆŒØ"ˆŒØ ˆŒØ 0ˆÕr%   r&   Údata_formatr)   c                 ó–   — t        |«      }d|vsd|vrt        d|j                  «       › �«      ‚|d   |d   f}t        |f||||dœ|¤ŽS )a�  
        Resize an image to `(size["height"], size["width"])`.

        Args:
            image (`np.ndarray`):
                Image to resize.
            size (`Dict[str, int]`):
                Dictionary in the format `{"height": int, "width": int}` specifying the size of the output image.
            resample (`PILImageResampling`, *optional*, defaults to `PILImageResampling.BILINEAR`):
                `PILImageResampling` filter to use when resizing the image e.g. `PILImageResampling.BILINEAR`.
            data_format (`ChannelDimension` or `str`, *optional*):
                The channel dimension format for the output image. If unset, the channel dimension format of the input
                image is used. Can be one of:
                - `"channels_first"` or `ChannelDimension.FIRST`: image in (num_channels, height, width) format.
                - `"channels_last"` or `ChannelDimension.LAST`: image in (height, width, num_channels) format.
                - `"none"` or `ChannelDimension.NONE`: image in (height, width) format.
            input_data_format (`ChannelDimension` or `str`, *optional*):
                The channel dimension format for the input image. If unset, the channel dimension format is inferred
                from the input image. Can be one of:
                - `"channels_first"` or `ChannelDimension.FIRST`: image in (num_channels, height, width) format.
                - `"channels_last"` or `ChannelDimension.LAST`: image in (height, width, num_channels) format.
                - `"none"` or `ChannelDimension.NONE`: image in (height, width) format.

        Returns:
            `np.ndarray`: The resized image.
        r"   r!   zFThe `size` dictionary must contain the keys `height` and `width`. Got )r3   rP   r\   r)   )r	   Ú
ValueErrorÚkeysr   )rY   r&   r3   rP   r\   r)   rZ   Úoutput_sizes           r#   r   zLayoutLMv2ImageProcessor.resize•   sy   € ôF ˜TÓ"ˆØ˜4Ñ 7°$Ñ#6ÜÐeÐfj×foÑfoÓfqÐerÐsÓtÐtØ˜H‘~ t¨G¡}Ð5ˆÜØð
àØØ#Ø/ñ
ð ñ
ð 	
r%   ÚimagesÚreturn_tensorsc           	      ó4  — |�|n| j                   }|�|n| j                  }t        |«      }|�|n| j                  }|�|n| j                  }|�|n| j
                  }|�|n| j                  }t        |«      }t        |«      st        d«      ‚t        |||¬«       |D �cg c]  }t        |«      ‘Œ }}|
€t        |d   «      }
|rKt        | d«       g }g }|D ]6  }t        ||||
¬«      \  }}|j                  |«       |j                  |«       Œ8 |r"|D �cg c]  }| j!                  ||||
¬«      ‘Œ }}|D �cg c]  }t#        ||
¬«      ‘Œ }}|D �cg c]  }t%        ||	|
¬«      ‘Œ }}t'        d|i|¬	«      }|r
|d
<   |d<   |S c c}w c c}w c c}w c c}w )a´  
        Preprocess an image or batch of images.

        Args:
            images (`ImageInput`):
                Image to preprocess.
            do_resize (`bool`, *optional*, defaults to `self.do_resize`):
                Whether to resize the image.
            size (`Dict[str, int]`, *optional*, defaults to `self.size`):
                Desired size of the output image after resizing.
            resample (`PILImageResampling`, *optional*, defaults to `self.resample`):
                Resampling filter to use if resizing the image. This can be one of the enum `PIL.Image` resampling
                filter. Only has an effect if `do_resize` is set to `True`.
            apply_ocr (`bool`, *optional*, defaults to `self.apply_ocr`):
                Whether to apply the Tesseract OCR engine to get words + normalized bounding boxes.
            ocr_lang (`str`, *optional*, defaults to `self.ocr_lang`):
                The language, specified by its ISO code, to be used by the Tesseract OCR engine. By default, English is
                used.
            tesseract_config (`str`, *optional*, defaults to `self.tesseract_config`):
                Any additional custom configuration flags that are forwarded to the `config` parameter when calling
                Tesseract.
            return_tensors (`str` or `TensorType`, *optional*):
                The type of tensors to return. Can be one of:
                    - Unset: Return a list of `np.ndarray`.
                    - `TensorType.TENSORFLOW` or `'tf'`: Return a batch of type `tf.Tensor`.
                    - `TensorType.PYTORCH` or `'pt'`: Return a batch of type `torch.Tensor`.
                    - `TensorType.NUMPY` or `'np'`: Return a batch of type `np.ndarray`.
                    - `TensorType.JAX` or `'jax'`: Return a batch of type `jax.numpy.ndarray`.
            data_format (`ChannelDimension` or `str`, *optional*, defaults to `ChannelDimension.FIRST`):
                The channel dimension format for the output image. Can be one of:
                    - `ChannelDimension.FIRST`: image in (num_channels, height, width) format.
                    - `ChannelDimension.LAST`: image in (height, width, num_channels) format.
        zkInvalid image type. Must be of type PIL.Image.Image, numpy.ndarray, torch.Tensor, tf.Tensor or jax.ndarray.)rO   r3   rP   r   r4   r,   )r&   r3   rP   r)   )Úinput_channel_dimrN   )r>   Útensor_typer?   Úboxes)rO   r3   r	   rP   rQ   rR   r(   r   r   r^   r   r   r   r   rK   r9   r   r
   r   r   )rY   ra   rO   r3   rP   rQ   rR   r(   rb   r\   r)   r&   Úwords_batchÚboxes_batchr?   rf   r>   s                    r#   Ú
preprocessz#LayoutLMv2ImageProcessor.preprocessÅ   sñ  € ð^ "+Ð!6‘I¸D¿N¹Nˆ	ØÐ'‰t¨T¯Y©YˆÜ˜TÓ"ˆØ'Ð3‘8¸¿¹ˆØ!*Ð!6‘I¸D¿N¹Nˆ	Ø'Ð3‘8¸¿¹ˆØ/?Ð/KÑ+ÐQU×QfÑQfÐä$ VÓ,ˆä˜FÔ#Üð:óð ô 	&ØØØõ	
ð 6<Ö<¨E”. Õ'Ð<ˆÐ<àÐ$ä >¸vÀa¹yÓ IÐáÜ˜d MÔ2ØˆKØˆKØò *�Ü.¨u°hÐ@PÐduÔv‘��uØ×"Ñ" 5Ô)Ø×"Ñ" 5Õ)ð*ñ
 ð $öàð —‘ %¨d¸XÐYj�ÕkðˆFð ð _eÖeÐUZÔ$ UÐ>OÖPÐeˆÐeàntö
ØejÔ'¨¨{ÐN_Ö`ð
ˆð 
ô  .°&Ð!9À~ÔVˆáØ'ˆD�‰MØ'ˆD�‰MØˆùòA =ùò ùò fùò
s   ÂFÄFÄ8FÅF)Ú__name__Ú
__module__Ú__qualname__Ú__doc__Úmodel_input_namesr   ÚBILINEARÚboolr   Ústrr   r   rX   ÚnpÚndarrayr   r   r   r   ÚFIRSTr   r   ÚPILÚImageri   Ú__classcell__)r[   s   @r#   rM   rM   e   sø  ø„ ñð. (Ð(Ðð Ø#Ø'9×'BÑ'BØØ"&Ø*,ñ1àð1ð �3˜�8‰nð1ð %ð	1ð
 ð1ð ˜3‘-ð1ð # 3™-ð1ð 
õ1ð4 (:×'BÑ'BØ>BØDHñ.
à�z‰zð.
ð �3˜�8‰nð.
ð %ð	.
ð
 ˜e CÐ)9Ð$9Ñ:Ñ;ð.
ð $ E¨#Ð/?Ð*?Ñ$@ÑAð.
ð 
�‰ó.
ñ` %Ó&ð %)Ø#Ø'+Ø$(Ø"&Ø*.Ø;?Ø(8×(>Ñ(>ØDHñdàðdð ˜D‘>ðdð �3˜�8‰nð	dð
 %ðdð ˜D‘>ðdð ˜3‘-ðdð # 3™-ðdð !  s¨J Ñ!7Ñ8ðdð &ðdð $ E¨#Ð/?Ð*?Ñ$@ÑAðdð 
�‰�‰òdó 'ôdr%   rM   )NN)+rm   Útypingr   r   r   Únumpyrr   Úimage_processing_utilsr   r   r	   Úimage_transformsr
   r   r   r   Úimage_utilsr   r   r   r   r   r   r   r   Úutilsr   r   r   r   r   r   ru   r4   Ú
get_loggerrj   Úloggerr$   rs   rq   rK   rM   Ú__all__rV   r%   r#   ú<module>r�      sÍ   ðñ ,ç (Ñ (ã ç UÑ Uß eÓ e÷	÷ 	ó 	÷÷ ñ ÔÛñ ÔÛà	ˆ×	Ñ	˜HÓ	%€òð '+Ø@Dñ	$#Ø�:‰:ð$#à
�3‰-ð$#ð ˜s‘mð$#ð    cÐ+;Ð&;Ñ <Ñ=ó	$#ôNEÐ1ô EðP &Ð
&�r%   