Ë
    T^(h_-  ã                   ó�   — d Z ddlmZmZmZ ddlZddlmZ ddl	m
Z
 ddlmZmZmZ  ej                  e«      Z G d„ d	e«      Zd	gZy)
z&
Feature extractor class for Wav2Vec2
é    )ÚListÚOptionalÚUnionNé   )ÚSequenceFeatureExtractor)ÚBatchFeature)ÚPaddingStrategyÚ
TensorTypeÚloggingc                   óh  ‡ — e Zd ZdZddgZ	 	 	 	 	 dˆ fd„	Ze	 ddeej                     deej                     de
deej                     fd„«       Z	 	 	 	 	 	 	 ddeej                  ee
   eej                     eee
      f   d	eeeef   d
ee   dedee   dee   deeeef      dee   defd„Zˆ xZS )ÚWav2Vec2FeatureExtractora  
    Constructs a Wav2Vec2 feature extractor.

    This feature extractor inherits from [`~feature_extraction_sequence_utils.SequenceFeatureExtractor`] which contains
    most of the main methods. Users should refer to this superclass for more information regarding those methods.

    Args:
        feature_size (`int`, *optional*, defaults to 1):
            The feature dimension of the extracted features.
        sampling_rate (`int`, *optional*, defaults to 16000):
            The sampling rate at which the audio files should be digitalized expressed in hertz (Hz).
        padding_value (`float`, *optional*, defaults to 0.0):
            The value that is used to fill the padding values.
        do_normalize (`bool`, *optional*, defaults to `True`):
            Whether or not to zero-mean unit-variance normalize the input. Normalizing can help to significantly
            improve the performance for some models, *e.g.*,
            [wav2vec2-lv60](https://huggingface.co/models?search=lv60).
        return_attention_mask (`bool`, *optional*, defaults to `False`):
            Whether or not [`~Wav2Vec2FeatureExtractor.__call__`] should return `attention_mask`.

            <Tip>

            Wav2Vec2 models that have set `config.feat_extract_norm == "group"`, such as
            [wav2vec2-base](https://huggingface.co/facebook/wav2vec2-base-960h), have **not** been trained using
            `attention_mask`. For such models, `input_values` should simply be padded with 0 and no `attention_mask`
            should be passed.

            For Wav2Vec2 models that have set `config.feat_extract_norm == "layer"`, such as
            [wav2vec2-lv60](https://huggingface.co/facebook/wav2vec2-large-960h-lv60-self), `attention_mask` should be
            passed for batched inference.

            </Tip>Úinput_valuesÚattention_maskc                 óH   •— t        ‰| �  d|||dœ|¤Ž || _        || _        y )N)Úfeature_sizeÚsampling_rateÚpadding_value© )ÚsuperÚ__init__Úreturn_attention_maskÚdo_normalize)Úselfr   r   r   r   r   ÚkwargsÚ	__class__s          €úv/var/www/skyplay_api_hub/venv/lib/python3.12/site-packages/transformers/models/wav2vec2/feature_extraction_wav2vec2.pyr   z!Wav2Vec2FeatureExtractor.__init__C   s0   ø€ ô 	‰ÑÐw lÀ-Ð_lÑwÐpvÒwØ%:ˆÔ"Ø(ˆÕó    r   Úreturnc                 ó  — |�³t        j                  |t         j                  «      }g }t        | |j	                  d«      «      D ]m  \  }}||d| j                  «       z
  t        j                  |d| j                  «       dz   «      z  }||j                  d   k  r|||d |j                  |«       Œo |S | D �cg c]<  }||j                  «       z
  t        j                  |j                  «       dz   «      z  ‘Œ> }}|S c c}w )z[
        Every array in the list is normalized to have zero mean and unit variance
        NéÿÿÿÿgH¯¼šò×z>r   )
ÚnpÚarrayÚint32ÚzipÚsumÚmeanÚsqrtÚvarÚshapeÚappend)r   r   r   Únormed_input_valuesÚvectorÚlengthÚnormed_sliceÚxs           r   Úzero_mean_unit_var_normz0Wav2Vec2FeatureExtractor.zero_mean_unit_var_normP   s  € ð Ð%ÜŸX™X n´b·h±hÓ?ˆNØ"$Ðä"% l°N×4FÑ4FÀrÓ4JÓ"Kò 9‘�˜Ø &¨°°¨×)=Ñ)=Ó)?Ñ ?Ä2Ç7Á7È6ÐRYÐSYÈ?×K^ÑK^ÓK`ÐcgÑKgÓChÑh�Ø˜L×.Ñ.¨qÑ1Ò1Ø,9�L  Ð)à#×*Ñ*¨<Õ8ð9ð #Ð"ð VbÖ"bÐPQ A¨¯©«¡L´B·G±G¸A¿E¹E»GÀd¹NÓ4KÓ#KÐ"bÐÐ"bà"Ð"ùò #cs   Â:AC?Ú
raw_speechÚpaddingÚ
max_lengthÚ
truncationÚpad_to_multiple_ofr   Úreturn_tensorsr   c	                 ó®  — |�;|| j                   k7  rYt        d| › d| j                   › d| j                   › d|› d�	«      ‚t        j                  d| j                  j
                  › d�«       t        |t        j                  «      xr t        |j                  «      d	kD  }
|
r&t        |j                  «      d
kD  rt        d| › �«      ‚|
xs@ t        |t        t        f«      xr( t        |d   t        j                  t        t        f«      }|s|g}t        d|i«      }| j                  ||||||¬«      }|d   }t        |d   t        j                  «      s8|D �cg c]'  }t        j                  |t        j                   ¬«      ‘Œ) c}|d<   �nt        |t        j                  «      s€t        |d   t        j                  «      rc|d   j"                  t        j"                  t        j$                  «      u r1|D �cg c]!  }|j'                  t        j                   «      ‘Œ# c}|d<   nkt        |t        j                  «      rQ|j"                  t        j"                  t        j$                  «      u r"|j'                  t        j                   «      |d<   |j)                  d«      }|�6|D �cg c]'  }t        j                  |t        j*                  ¬«      ‘Œ) c}|d<   | j,                  rK| j/                  ||¬«      t0        j2                  ur|nd}| j5                  |d   || j6                  ¬«      |d<   |�|j9                  |«      }|S c c}w c c}w c c}w )aà  
        Main method to featurize and prepare for the model one or several sequence(s).

        Args:
            raw_speech (`np.ndarray`, `List[float]`, `List[np.ndarray]`, `List[List[float]]`):
                The sequence or batch of sequences to be padded. Each sequence can be a numpy array, a list of float
                values, a list of numpy arrays or a list of list of float values. Must be mono channel audio, not
                stereo, i.e. single float per timestep.
            padding (`bool`, `str` or [`~utils.PaddingStrategy`], *optional*, defaults to `False`):
                Select a strategy to pad the returned sequences (according to the model's padding side and padding
                index) among:

                - `True` or `'longest'`: Pad to the longest sequence in the batch (or no padding if only a single
                  sequence if provided).
                - `'max_length'`: Pad to a maximum length specified with the argument `max_length` or to the maximum
                  acceptable input length for the model if that argument is not provided.
                - `False` or `'do_not_pad'` (default): No padding (i.e., can output a batch with sequences of different
                  lengths).
            max_length (`int`, *optional*):
                Maximum length of the returned list and optionally padding length (see above).
            truncation (`bool`):
                Activates truncation to cut input sequences longer than *max_length* to *max_length*.
            pad_to_multiple_of (`int`, *optional*):
                If set will pad the sequence to a multiple of the provided value.

                This is especially useful to enable the use of Tensor Cores on NVIDIA hardware with compute capability
                `>= 7.5` (Volta), or on TPUs which benefit from having sequence lengths be a multiple of 128.
            return_attention_mask (`bool`, *optional*):
                Whether to return the attention mask. If left to the default, will return the attention mask according
                to the specific feature_extractor's default.

                [What are attention masks?](../glossary#attention-mask)

                <Tip>

                Wav2Vec2 models that have set `config.feat_extract_norm == "group"`, such as
                [wav2vec2-base](https://huggingface.co/facebook/wav2vec2-base-960h), have **not** been trained using
                `attention_mask`. For such models, `input_values` should simply be padded with 0 and no
                `attention_mask` should be passed.

                For Wav2Vec2 models that have set `config.feat_extract_norm == "layer"`, such as
                [wav2vec2-lv60](https://huggingface.co/facebook/wav2vec2-large-960h-lv60-self), `attention_mask` should
                be passed for batched inference.

                </Tip>

            return_tensors (`str` or [`~utils.TensorType`], *optional*):
                If set, will return tensors instead of list of python integers. Acceptable values are:

                - `'tf'`: Return TensorFlow `tf.constant` objects.
                - `'pt'`: Return PyTorch `torch.Tensor` objects.
                - `'np'`: Return Numpy `np.ndarray` objects.
            sampling_rate (`int`, *optional*):
                The sampling rate at which the `raw_speech` input was sampled. It is strongly recommended to pass
                `sampling_rate` at the forward call to prevent silent errors.
            padding_value (`float`, *optional*, defaults to 0.0):
        Nz3The model corresponding to this feature extractor: z& was trained using a sampling rate of zI. Please make sure that the provided `raw_speech` input was sampled with z	 and not ú.zDIt is strongly recommended to pass the `sampling_rate` argument to `zN()`. Failing to do so can result in silent errors that might be hard to debug.é   é   z2Only mono-channel audio is supported for input to r   r   )r2   r3   r4   r5   r   )Údtyper   )r3   )r   r   )r   Ú
ValueErrorÚloggerÚwarningr   Ú__name__Ú
isinstancer!   ÚndarrayÚlenr)   ÚlistÚtupler   ÚpadÚasarrayÚfloat32r;   Úfloat64ÚastypeÚgetr#   r   Ú_get_padding_strategiesr	   Ú
DO_NOT_PADr0   r   Úconvert_to_tensors)r   r1   r2   r3   r4   r5   r   r6   r   r   Úis_batched_numpyÚ
is_batchedÚencoded_inputsÚpadded_inputsr   r"   r   s                    r   Ú__call__z!Wav2Vec2FeatureExtractor.__call__f   s  € ðL Ð$Ø × 2Ñ 2Ò2Ü ØIÈ$Èð PØ×*Ñ*Ð+ð ,Ø×*Ñ*Ð+¨9°]°OÀ1ðFóð ô �N‰NØVÐW[×WeÑWe×WnÑWnÐVoð p\ð \ôô
 & j´"·*±*Ó=Ò[Ä#Àj×FVÑFVÓBWÐZ[ÑB[ÐÙ¤ J×$4Ñ$4Ó 5¸Ò 9ÜÐQÐRVÐQWÐXÓYÐYØ%ò 
Ü�z¤D¬% =Ó1Òl´zÀ*ÈQÁ-ÔRT×R\ÑR\Ô^cÔeiÐQjÓ7kð 	ñ
 Ø$˜ˆJô & ~°zÐ&BÓCˆàŸ™ØØØ!Ø!Ø1Ø"7ð !ó 
ˆð % ^Ñ4ˆÜ˜, q™/¬2¯:©:Ô6Ø^jÖ,kÐUZ¬R¯Z©Z¸ÄRÇZÁZÖ-PÒ,kˆM˜.Ó)ä˜<¬¯©Ô4Ü˜<¨™?¬B¯J©JÔ7Ø˜Q‘×%Ñ%¬¯©´"·*±*Ó)=Ñ=àS_Ö,`È%¨U¯\©\¼"¿*¹*Õ-EÒ,`ˆM˜.Ò)Ü˜¤b§j¡jÔ1°l×6HÑ6HÌBÏHÉHÔUW×U_ÑU_ÓL`Ñ6`Ø,8×,?Ñ,?ÄÇ
Á
Ó,KˆM˜.Ñ)ð '×*Ñ*Ð+;Ó<ˆØÐ%Ø^lÖ.mÐUZ¬r¯z©z¸%ÄrÇxÁxÖ/PÒ.mˆMÐ*Ñ+ð ×Òð ×/Ñ/°ÀJÐ/ÓOÔWf×WqÑWqÑqñ àð ð
 -1×,HÑ,HØ˜nÑ-¸nÐ\`×\nÑ\nð -Ió -ˆM˜.Ñ)ð Ð%Ø)×<Ñ<¸^ÓLˆMàÐùò; -lùò -aùò /ns   Å,MÇ;&MÊ*,M)r9   i€>  ç        FT)rS   )FNFNNNN)r?   Ú
__module__Ú__qualname__Ú__doc__Úmodel_input_namesr   Ústaticmethodr   r!   rA   Úfloatr0   r   ÚboolÚstrr	   r   Úintr
   r   rR   Ú__classcell__)r   s   @r   r   r      sO  ø„ ñðB (Ð)9Ð:Ðð ØØØ#Øõ)ð àadñ#Ø˜2Ÿ:™:Ñ&ð#Ø8<¸R¿Z¹ZÑ8Hð#ØY^ð#à	ˆb�j‰jÑ	ò#ó ð#ð0 6;Ø$(Ø Ø,0Ø04Ø;?Ø'+ñJà˜"Ÿ*™* d¨5¡k°4¸¿
¹
Ñ3CÀTÈ$ÈuÉ+ÑEVÐVÑWðJð �t˜S /Ð1Ñ2ðJð ˜S‘Mð	Jð
 ðJð % S™MðJð  (¨™~ðJð !  s¨J Ñ!7Ñ8ðJð   ‘}ðJð 
÷Jr   r   )rV   Útypingr   r   r   Únumpyr!   Ú!feature_extraction_sequence_utilsr   Úfeature_extraction_utilsr   Úutilsr	   r
   r   Ú
get_loggerr?   r=   r   Ú__all__r   r   r   ú<module>re      sO   ðñ÷ )Ñ (ã å IÝ 4ß 9Ñ 9ð 
ˆ×	Ñ	˜HÓ	%€ôQÐ7ô Qðh &Ð
&�r   