Ë
    S^(h+  ã                   ó>   — d Z ddlZddlmZ ddlmZ  G d„ de«      Zy)z$
Speech processor class for M-CTC-T
é    N)Úcontextmanageré   )ÚProcessorMixinc                   óR   ‡ — e Zd ZdZdZdZˆ fd„Zd„ Zd„ Zd„ Z	d„ Z
ed	„ «       Zˆ xZS )
ÚMCTCTProcessora[  
    Constructs a MCTCT processor which wraps a MCTCT feature extractor and a MCTCT tokenizer into a single processor.

    [`MCTCTProcessor`] offers all the functionalities of [`MCTCTFeatureExtractor`] and [`AutoTokenizer`]. See the
    [`~MCTCTProcessor.__call__`] and [`~MCTCTProcessor.decode`] for more information.

    Args:
        feature_extractor (`MCTCTFeatureExtractor`):
            An instance of [`MCTCTFeatureExtractor`]. The feature extractor is a required input.
        tokenizer (`AutoTokenizer`):
            An instance of [`AutoTokenizer`]. The tokenizer is a required input.
    ÚMCTCTFeatureExtractorÚAutoTokenizerc                 óV   •— t         ‰| �  ||«       | j                  | _        d| _        y )NF)ÚsuperÚ__init__Úfeature_extractorÚcurrent_processorÚ_in_target_context_manager)Úselfr   Ú	tokenizerÚ	__class__s      €ús/var/www/skyplay_api_hub/venv/lib/python3.12/site-packages/transformers/models/deprecated/mctct/processing_mctct.pyr   zMCTCTProcessor.__init__*   s)   ø€ Ü‰ÑÐ*¨IÔ6Ø!%×!7Ñ!7ˆÔØ*/ˆÕ'ó    c                 óÐ  — | j                   r | j                  |i |¤ŽS d|v r't        j                  d«       |j	                  d«      }n|j	                  dd«      }|j	                  dd«      }|j	                  dd«      }t        |«      dkD  r
|d   }|dd }|€|€t        d	«      ‚|� | j                  |g|¢­d|i|¤Ž}|� | j                  |fi |¤Ž}|€S |€S d
   d<   |S )a¤  
        When used in normal mode, this method forwards all its arguments to MCTCTFeatureExtractor's
        [`~MCTCTFeatureExtractor.__call__`] and returns its output. If used in the context
        [`~MCTCTProcessor.as_target_processor`] this method forwards all its arguments to AutoTokenizer's
        [`~AutoTokenizer.__call__`]. Please refer to the docstring of the above two methods for more information.
        Ú
raw_speechzLUsing `raw_speech` as a keyword argument is deprecated. Use `audio` instead.ÚaudioNÚsampling_rateÚtextr   é   zAYou need to specify either an `audio` or `text` input to process.Ú	input_idsÚlabels)	r   r   ÚwarningsÚwarnÚpopÚlenÚ
ValueErrorr   r   )r   ÚargsÚkwargsr   r   r   ÚinputsÚ	encodingss           r   Ú__call__zMCTCTProcessor.__call__/   s  € ð ×*Ò*Ø)�4×)Ñ)¨4Ð:°6Ñ:Ð:à˜6Ñ!Ü�M‰MÐhÔiØ—J‘J˜|Ó,‰Eà—J‘J˜w¨Ó-ˆEØŸ
™
 ?°DÓ9ˆØ�z‰z˜& $Ó'ˆÜˆt‹9�qŠ=Ø˜‘GˆEØ˜˜�8ˆDàˆ=˜T˜\ÜÐ`ÓaÐaàÐØ+�T×+Ñ+¨EÐ`°DÒ`ÈÐ`ÐY_Ñ`ˆFØÐØ&˜Ÿ™ tÑ6¨vÑ6ˆIàˆ<ØˆMØˆ]ØÐà(¨Ñ5ˆF�8ÑØˆMr   c                 ó:   —  | j                   j                  |i |¤ŽS )z½
        This method forwards all its arguments to AutoTokenizer's [`~PreTrainedTokenizer.batch_decode`]. Please refer
        to the docstring of this method for more information.
        )r   Úbatch_decode©r   r"   r#   s      r   r(   zMCTCTProcessor.batch_decodeU   s    € ð
 +ˆt�~‰~×*Ñ*¨DÐ;°FÑ;Ð;r   c                 óp  — | j                   r | j                  j                  |i |¤ŽS |j                  dd«      }|j                  dd«      }t	        |«      dkD  r
|d   }|dd }|�  | j
                  j                  |g|¢­i |¤Ž}|� | j                  j                  |fi |¤Ž}|€|S |€|S |d   |d<   |S )a¦  
        When used in normal mode, this method forwards all its arguments to MCTCTFeatureExtractor's
        [`~MCTCTFeatureExtractor.pad`] and returns its output. If used in the context
        [`~MCTCTProcessor.as_target_processor`] this method forwards all its arguments to PreTrainedTokenizer's
        [`~PreTrainedTokenizer.pad`]. Please refer to the docstring of the above two methods for more information.
        Úinput_featuresNr   r   r   r   )r   r   Úpadr   r    r   r   )r   r"   r#   r+   r   s        r   r,   zMCTCTProcessor.pad\   sà   € ð ×*Ò*Ø-�4×)Ñ)×-Ñ-¨tÐ>°vÑ>Ð>àŸ™Ð$4°dÓ;ˆØ—‘˜H dÓ+ˆÜˆt‹9�qŠ=Ø! !™WˆNØ˜˜�8ˆDàÐ%Ø7˜T×3Ñ3×7Ñ7¸ÐXÈÒXÐQWÑXˆNØÐØ'�T—^‘^×'Ñ'¨Ñ9°&Ñ9ˆFàˆ>Ø!Ð!ØÐ#ØˆMà'-¨kÑ':ˆN˜8Ñ$Ø!Ð!r   c                 ó:   —  | j                   j                  |i |¤ŽS )z·
        This method forwards all its arguments to AutoTokenizer's [`~PreTrainedTokenizer.decode`]. Please refer to the
        docstring of this method for more information.
        )r   Údecoder)   s      r   r.   zMCTCTProcessor.decodez   s    € ð
 %ˆt�~‰~×$Ñ$ dÐ5¨fÑ5Ð5r   c              #   óž   K  — t        j                  d«       d| _        | j                  | _        d–— | j
                  | _        d| _        y­w)z�
        Temporarily sets the tokenizer for processing the input. Useful for encoding the labels when fine-tuning MCTCT.
        zî`as_target_processor` is deprecated and will be removed in v5 of Transformers. You can process your labels by using the argument `text` of the regular `__call__` method (either in the same call as your audio inputs, or in a separate call.TNF)r   r   r   r   r   r   )r   s    r   Úas_target_processorz"MCTCTProcessor.as_target_processor�   sH   è ø€ ô
 	�‰ð8ô	
ð
 +/ˆÔ'Ø!%§¡ˆÔÛØ!%×!7Ñ!7ˆÔØ*/ˆÕ'ùs   ‚AA)Ú__name__Ú
__module__Ú__qualname__Ú__doc__Úfeature_extractor_classÚtokenizer_classr   r&   r(   r,   r.   r   r0   Ú__classcell__)r   s   @r   r   r      sC   ø„ ñð 6ÐØ%€Oô0ò
$òL<ò"ò<6ð ñ0ó ô0r   r   )r4   r   Ú
contextlibr   Úprocessing_utilsr   r   © r   r   ú<module>r;      s#   ðñó Ý %å /ôv0�^õ v0r   