Ë
    S^(hù*  ã                   ó    — d Z ddlmZmZmZ ddlZddlmZm	Z	m
Z
 ddlmZ ddlmZ ddlmZmZ  ej$                  e«      Z G d	„ d
e«      Zd
gZy)z"
Feature extractor class for CLVP
é    )ÚListÚOptionalÚUnionNé   )Úmel_filter_bankÚspectrogramÚwindow_function)ÚSequenceFeatureExtractor)ÚBatchFeature)Ú
TensorTypeÚloggingc                   ó.  ‡ — e Zd ZdZddgZ	 	 	 	 	 	 	 	 	 dˆ fd„	Zdej                  dej                  fd„Z		 	 	 	 	 	 	 dd	e
ej                  ee   eej                     eee      f   d
ee   dedee   dee
eef      dee   dee   dee   defd„Zˆ xZS )ÚClvpFeatureExtractora!  
    Constructs a CLVP feature extractor.

    This feature extractor inherits from [`~feature_extraction_sequence_utils.SequenceFeatureExtractor`] which contains
    most of the main methods. Users should refer to this superclass for more information regarding those methods.

    This class extracts log-mel-spectrogram features from raw speech using a custom numpy implementation of the `Short
    Time Fourier Transform` which should match pytorch's `torch.stft` equivalent.

    Args:
        feature_size (`int`, *optional*, defaults to 80):
            The feature dimension of the extracted features.
        sampling_rate (`int`, *optional*, defaults to 22050):
            The sampling rate at which the audio files should be digitalized expressed in hertz (Hz).
        default_audio_length (`int`, *optional*, defaults to 6):
            The default length of raw audio in seconds. If `max_length` is not set during `__call__` then it will
            automatically be set to default_audio_length * `self.sampling_rate`.
        hop_length (`int`, *optional*, defaults to 256):
            Length of the overlaping windows for the STFT used to obtain the Mel Frequency coefficients.
        chunk_length (`int`, *optional*, defaults to 30):
            The maximum number of chuncks of `sampling_rate` samples used to trim and pad longer or shorter audio
            sequences.
        n_fft (`int`, *optional*, defaults to 1024):
            Size of the Fourier transform.
        padding_value (`float`, *optional*, defaults to 0.0):
            Padding value used to pad the audio. Should correspond to silences.
        mel_norms (`list` of length `feature_size`, *optional*):
            If `mel_norms` is provided then it will be used to normalize the log-mel spectrograms along each
            mel-filter.
        return_attention_mask (`bool`, *optional*, defaults to `False`):
            Whether to return the attention mask. If left to the default, it will return the attention mask.

            [What are attention masks?](../glossary#attention-mask)
    Úinput_featuresÚattention_maskc
           	      óø   •— t        ‰| �  d	||||	dœ|
¤Ž || _        || _        || _        ||z  | _        | j
                  |z  | _        || _        || _        || _	        t        d|dz  z   |dd|dd¬«      | _        y )
N)Úfeature_sizeÚsampling_rateÚpadding_valueÚreturn_attention_maské   é   ç        g     @¿@ÚslaneyÚhtk)Únum_frequency_binsÚnum_mel_filtersÚmin_frequencyÚmax_frequencyr   ÚnormÚ	mel_scale© )ÚsuperÚ__init__Ún_fftÚ
hop_lengthÚchunk_lengthÚ	n_samplesÚnb_max_framesr   Údefault_audio_lengthÚ	mel_normsr   Úmel_filters)Úselfr   r   r*   r&   r'   r%   r   r+   r   ÚkwargsÚ	__class__s              €ún/var/www/skyplay_api_hub/venv/lib/python3.12/site-packages/transformers/models/clvp/feature_extraction_clvp.pyr$   zClvpFeatureExtractor.__init__G   s¤   ø€ ô 	‰Ñð 	
Ø%Ø'Ø'Ø"7ñ		
ð
 ò	
ð ˆŒ
Ø$ˆŒØ(ˆÔØ%¨Ñ5ˆŒØ!Ÿ^™^¨zÑ9ˆÔØ*ˆÔØ$8ˆÔ!Ø"ˆŒÜ*Ø  E¨Q¡JÑ/Ø(ØØ Ø'ØØô
ˆÕó    ÚwaveformÚreturnc           	      óN  — t        |t        | j                  d«      | j                  | j                  d| j                  d¬«      }t        j                  t        j                  |dd¬«      «      }| j                  �)|t        j                  | j                  «      dd…df   z  }|S )z¸
        This method first computes the log-mel spectrogram of the provided audio then applies normalization along the
        each mel-filterbank, if `mel_norms` is provided.
        Úhanng       @N)Úframe_lengthr&   Úpowerr,   Úlog_melgñhãˆµøä>)Úa_minÚa_max)
r   r	   r%   r&   r,   ÚnpÚlogÚclipr+   Úarray)r-   r2   Úlog_specs      r0   Ú_np_extract_fbank_featuresz/ClvpFeatureExtractor._np_extract_fbank_featuresm   sˆ   € ô
 ØÜ˜DŸJ™J¨Ó/ØŸ™Ø—‘ØØ×(Ñ(Øô
ˆô —6‘6œ"Ÿ'™' (°$¸dÔCÓDˆà�>‰>Ð%Ø¤"§(¡(¨4¯>©>Ó":º1¸d¸7Ñ"CÑCˆHàˆr1   Ú
max_lengthÚ
raw_speechr   Ú
truncationÚpad_to_multiple_ofÚreturn_tensorsr   Úpaddingc	                 óX  — |�O|| j                   k7  rmt        d| j                  j                  › d| j                   › d| j                   › d|› d�	«      ‚t        j                  d| j                  j                  › d�«       t        |t        j                  «      xr t        |j                  «      dkD  }
|
r&t        |j                  «      d	kD  rt        d
| › �«      ‚|
xs@ t        |t        t        f«      xr( t        |d   t        j                  t        t        f«      }|r>|D �cg c]2  }t        j                  |gt        j                  ¬«      j                  ‘Œ4 }}nª|s@t        |t        j                  «      s&t        j                  |t        j                  ¬«      }nht        |t        j                  «      rN|j                   t        j                   t        j"                  «      u r|j%                  t        j                  «      }|s!t        j                  |g«      j                  g}t'        d|i«      }|€| j(                  | j                   z  n|}| j+                  ||||||¬«      }|j-                  d«      j/                  d	dd«      }|d   D �cg c]0  }| j1                  |«      j%                  t        j                  «      ‘Œ2 }}t        |d   t2        «      r'|D �cg c]  }t        j                  |«      ‘Œ c}|d<   n||d<   |j5                  |«      S c c}w c c}w c c}w )aô	  
        `ClvpFeatureExtractor` is used to extract various voice specific properties such as the pitch and tone of the
        voice, speaking speed, and even speaking defects like a lisp or stuttering from a sample voice or `raw_speech`.

        First the voice is padded or truncated in a way such that it becomes a waveform of `self.default_audio_length`
        seconds long and then the log-mel spectrogram is extracted from it.

        Args:
            raw_speech (`np.ndarray`, `List[float]`, `List[np.ndarray]`, `List[List[float]]`):
                The sequence or batch of sequences to be padded. Each sequence can be a numpy array, a list of float
                values, a list of numpy arrays or a list of list of float values. Must be mono channel audio, not
                stereo, i.e. single float per timestep.
            sampling_rate (`int`, *optional*):
                The sampling rate at which the `raw_speech` input was sampled. It is strongly recommended to pass
                `sampling_rate` at the forward call to prevent silent errors and allow automatic speech recognition
                pipeline.
            truncation (`bool`, *optional*, default to `True`):
                Activates truncation to cut input sequences longer than *max_length* to *max_length*.
            pad_to_multiple_of (`int`, *optional*):
                If set will pad the sequence to a multiple of the provided value.

                This is especially useful to enable the use of Tensor Cores on NVIDIA hardware with compute capability
                `>= 7.5` (Volta), or on TPUs which benefit from having sequence lengths be a multiple of 128.
            return_attention_mask (`bool`, *optional*, defaults to `True`):
                Whether to return the attention mask. If left to the default, it will return the attention mask.

                [What are attention masks?](../glossary#attention-mask)
            return_tensors (`str` or [`~utils.TensorType`], *optional*):
                If set, will return tensors instead of list of python integers. Acceptable values are:

                - `'tf'`: Return TensorFlow `tf.constant` objects.
                - `'pt'`: Return PyTorch `torch.Tensor` objects.
                - `'np'`: Return Numpy `np.ndarray` objects.
            padding_value (`float`, *optional*, defaults to 0.0):
                The value that is used to fill the padding values / vectors.
            max_length (`int`, *optional*):
                The maximum input length of the inputs.
        z3The model corresponding to this feature extractor: z& was trained using a sampling rate of zI. Please make sure that the provided `raw_speech` input was sampled with z	 and not ú.zDIt is strongly recommended to pass the `sampling_rate` argument to `zN()`. Failing to do so can result in silent errors that might be hard to debug.r   r   z2Only mono-channel audio is supported for input to r   )Údtyper   )rF   rA   rC   rD   r   )r   Ú
ValueErrorr/   Ú__name__ÚloggerÚwarningÚ
isinstancer;   ÚndarrayÚlenÚshapeÚlistÚtupleÚasarrayÚfloat32ÚTrI   Úfloat64Úastyper   r*   ÚpadÚgetÚ	transposer@   r   Úconvert_to_tensors)r-   rB   r   rC   rD   rE   r   rF   rA   r.   Úis_batched_numpyÚ
is_batchedÚspeechÚbatched_speechÚpadded_inputsr   r2   Úfeatures                     r0   Ú__call__zClvpFeatureExtractor.__call__ƒ   sé  € ðf Ð$Ø × 2Ñ 2Ò2Ü ØIÈ$Ï.É.×JaÑJaÐIbð c)Ø)-×);Ñ);Ð(<ð =)Ø)-×);Ñ);Ð(<¸IÀmÀ_ÐTUðWóð ô �N‰NØVÐW[×WeÑWe×WnÑWnÐVoð p\ð \ôô
 & j´"·*±*Ó=Ò[Ä#Àj×FVÑFVÓBWÐZ[ÑB[ÐÙ¤ J×$4Ñ$4Ó 5¸Ò 9ÜÐQÐRVÐQWÐXÓYÐYØ%ò 
Ü�z¤D¬% =Ó1Òl´zÀ*ÈQÁ-ÔRT×R\ÑR\Ô^cÔeiÐQjÓ7kð 	ñ ØQ[Ö\Àvœ"Ÿ*™* f X´R·Z±ZÔ@×BÓBÐ\ˆJÑ\Ù¤J¨z¼2¿:¹:Ô$FÜŸ™ J´b·j±jÔA‰JÜ˜
¤B§J¡JÔ/°J×4DÑ4DÌÏÉÔQS×Q[ÑQ[ÓH\Ñ4\Ø#×*Ñ*¬2¯:©:Ó6ˆJñ ÜŸ*™* j \Ó2×4Ñ4Ð5ˆJä%Ð'7¸Ð&DÓEˆàGQÐGY�T×.Ñ.°×1CÑ1CÒCÐ_iˆ
àŸ™ØØØ!Ø!Ø1Ø"7ð !ó 
ˆð '×*Ñ*Ð+;Ó<×FÑFÀqÈ!ÈQÓOˆð ZhÐhiÑYjö
ØMUˆD×+Ñ+¨HÓ5×<Ñ<¼R¿Z¹ZÕHð
ˆð 
ô �n QÑ'¬Ô.ØR`Ö.aÀw¬r¯z©z¸'Õ/BÒ.aˆMÐ*Ò+à.<ˆMÐ*Ñ+à×/Ñ/°Ó?Ð?ùòG ]ùò4
ùò
 /bs   Ä%7LÊ5L"Ë$L')	éP   i"V  é   é   é   i   r   NF)NTNNTrA   N)rK   Ú
__module__Ú__qualname__Ú__doc__Úmodel_input_namesr$   r;   r>   rO   r@   r   r   Úfloatr   ÚintÚboolÚstrr   r   rc   Ú__classcell__)r/   s   @r0   r   r   !   s'  ø„ ñ!ðF *Ð+;Ð<Ðð ØØØØØØØØ#õ$
ðL°2·8±8ð ÀÇ
Á
ó ð2 (,ØØ,0Ø;?Ø04Ø!-Ø$(ñk@à˜"Ÿ*™* d¨5¡k°4¸¿
¹
Ñ3CÀTÈ$ÈuÉ+ÑEVÐVÑWðk@ð   ‘}ðk@ð ð	k@ð
 % S™Mðk@ð !  s¨J Ñ!7Ñ8ðk@ð  (¨™~ðk@ð ˜#‘ðk@ð ˜S‘Mðk@ð 
÷k@r1   r   )rj   Útypingr   r   r   Únumpyr;   Úaudio_utilsr   r   r	   Ú!feature_extraction_sequence_utilsr
   Úfeature_extraction_utilsr   Úutilsr   r   Ú
get_loggerrK   rL   r   Ú__all__r"   r1   r0   ú<module>ry      sT   ðñ ÷ )Ñ (ã ç HÑ HÝ IÝ 4ß (ð 
ˆ×	Ñ	˜HÓ	%€ôM@Ð3ô M@ð` "Ð
"�r1   