Ë
    T^(hÉE  ã                   ó¸   — d Z ddlZddlmZmZmZmZmZ ddlZ	ddl
mZmZmZmZ ddlmZ ddlmZ ddlmZmZmZ  ej.                  e«      Z G d	„ d
e«      Zd
gZy)z%Feature extractor class for SpeechT5.é    N)ÚAnyÚDictÚListÚOptionalÚUnioné   )Úmel_filter_bankÚoptimal_fft_lengthÚspectrogramÚwindow_function)ÚSequenceFeatureExtractor)ÚBatchFeature)ÚPaddingStrategyÚ
TensorTypeÚloggingc                   ó   ‡ — e Zd ZdZddgZ	 	 	 	 	 	 	 	 	 	 	 	 	 	 d#dededededed	ed
edededededededefˆ fd„Z	e
	 d$deej                     deej                     dedeej                     fd„«       Zdej                  dej                  fd„Z	 	 	 	 	 	 	 	 	 d%deeej                  ee   eej                     eee      f      deeej                  ee   eej                     eee      f      deeeef   dee   dedee   dee   deeeef      dee   defd„Z	 	 	 	 	 	 	 d&deej                  ee   eej                     eee      f   d edeeeef   dee   dedee   dee   deeeef      defd!„Zdeeef   fˆ fd"„Zˆ xZS )'ÚSpeechT5FeatureExtractora
  
    Constructs a SpeechT5 feature extractor.

    This class can pre-process a raw speech signal by (optionally) normalizing to zero-mean unit-variance, for use by
    the SpeechT5 speech encoder prenet.

    This class can also extract log-mel filter bank features from raw speech, for use by the SpeechT5 speech decoder
    prenet.

    This feature extractor inherits from [`~feature_extraction_sequence_utils.SequenceFeatureExtractor`] which contains
    most of the main methods. Users should refer to this superclass for more information regarding those methods.

    Args:
        feature_size (`int`, *optional*, defaults to 1):
            The feature dimension of the extracted features.
        sampling_rate (`int`, *optional*, defaults to 16000):
            The sampling rate at which the audio files should be digitalized expressed in hertz (Hz).
        padding_value (`float`, *optional*, defaults to 0.0):
            The value that is used to fill the padding values.
        do_normalize (`bool`, *optional*, defaults to `False`):
            Whether or not to zero-mean unit-variance normalize the input. Normalizing can help to significantly
            improve the performance for some models.
        num_mel_bins (`int`, *optional*, defaults to 80):
            The number of mel-frequency bins in the extracted spectrogram features.
        hop_length (`int`, *optional*, defaults to 16):
            Number of ms between windows. Otherwise referred to as "shift" in many papers.
        win_length (`int`, *optional*, defaults to 64):
            Number of ms per window.
        win_function (`str`, *optional*, defaults to `"hann_window"`):
            Name for the window function used for windowing, must be accessible via `torch.{win_function}`
        frame_signal_scale (`float`, *optional*, defaults to 1.0):
            Constant multiplied in creating the frames before applying DFT. This argument is deprecated.
        fmin (`float`, *optional*, defaults to 80):
            Minimum mel frequency in Hz.
        fmax (`float`, *optional*, defaults to 7600):
            Maximum mel frequency in Hz.
        mel_floor (`float`, *optional*, defaults to 1e-10):
            Minimum value of mel frequency banks.
        reduction_factor (`int`, *optional*, defaults to 2):
            Spectrogram length reduction factor. This argument is deprecated.
        return_attention_mask (`bool`, *optional*, defaults to `True`):
            Whether or not [`~SpeechT5FeatureExtractor.__call__`] should return `attention_mask`.
    Úinput_valuesÚattention_maskÚfeature_sizeÚsampling_rateÚpadding_valueÚdo_normalizeÚnum_mel_binsÚ
hop_lengthÚ
win_lengthÚwin_functionÚframe_signal_scaleÚfminÚfmaxÚ	mel_floorÚreduction_factorÚreturn_attention_maskc           	      óº  •— t        ‰| �  d|||dœ|¤Ž || _        || _        || _        || _        || _        || _        |	| _        |
| _	        || _
        || _        || _        ||z  dz  | _        ||z  dz  | _        t        | j                  «      | _        | j                   dz  dz   | _        t%        | j                  | j                  d¬«      | _        t)        | j"                  | j                  | j                  | j                  | j*                  dd¬«      | _        |	d	k7  rt/        j0                  d
t2        «       |dk7  rt/        j0                  dt2        «       y y )N)r   r   r   iè  é   é   T)Úwindow_lengthÚnameÚperiodicÚslaney)Únum_frequency_binsÚnum_mel_filtersÚmin_frequencyÚmax_frequencyr   ÚnormÚ	mel_scaleç      ð?zeThe argument `frame_signal_scale` is deprecated and will be removed in version 4.30.0 of Transformersg       @zcThe argument `reduction_factor` is deprecated and will be removed in version 4.30.0 of Transformers© )ÚsuperÚ__init__r   r#   r   r   r   r   r   r   r    r!   r"   Úsample_sizeÚsample_strider
   Ún_fftÚn_freqsr   Úwindowr	   r   Úmel_filtersÚwarningsÚwarnÚFutureWarning)Úselfr   r   r   r   r   r   r   r   r   r   r    r!   r"   r#   ÚkwargsÚ	__class__s                   €úv/var/www/skyplay_api_hub/venv/lib/python3.12/site-packages/transformers/models/speecht5/feature_extraction_speecht5.pyr4   z!SpeechT5FeatureExtractor.__init__N   sQ  ø€ ô$ 	‰ÑÐw lÀ-Ð_lÑwÐpvÒwØ(ˆÔØ%:ˆÔ"à(ˆÔØ$ˆŒØ$ˆŒØ(ˆÔØ"4ˆÔØˆŒ	ØˆŒ	Ø"ˆŒØ 0ˆÔà%¨Ñ5¸Ñ=ˆÔØ'¨-Ñ7¸4Ñ?ˆÔÜ'¨×(8Ñ(8Ó9ˆŒ
ØŸ
™
 a™¨1Ñ,ˆŒä%°D×4DÑ4DÈ4×K\ÑK\ÐgkÔlˆŒä*Ø#Ÿ|™|Ø ×-Ñ-ØŸ)™)ØŸ)™)Ø×,Ñ,ØØô
ˆÔð  Ò$Ü�M‰MØwÜôð ˜sÒ"Ü�M‰MØuÜõð #ó    Úreturnc                 ó  — |�³t        j                  |t         j                  «      }g }t        | |j	                  d«      «      D ]m  \  }}||d| j                  «       z
  t        j                  |d| j                  «       dz   «      z  }||j                  d   k  r|||d |j                  |«       Œo |S | D �cg c]<  }||j                  «       z
  t        j                  |j                  «       dz   «      z  ‘Œ> }}|S c c}w )z[
        Every array in the list is normalized to have zero mean and unit variance
        NéÿÿÿÿgH¯¼šò×z>r   )
ÚnpÚarrayÚint32ÚzipÚsumÚmeanÚsqrtÚvarÚshapeÚappend)r   r   r   Únormed_input_valuesÚvectorÚlengthÚnormed_sliceÚxs           rA   Úzero_mean_unit_var_normz0SpeechT5FeatureExtractor.zero_mean_unit_var_normŠ   s  € ð Ð%ÜŸX™X n´b·h±hÓ?ˆNØ"$Ðä"% l°N×4FÑ4FÀrÓ4JÓ"Kò 9‘�˜Ø &¨°°¨×)=Ñ)=Ó)?Ñ ?Ä2Ç7Á7È6ÐRYÐSYÈ?×K^ÑK^ÓK`ÐcgÑKgÓChÑh�Ø˜L×.Ñ.¨qÑ1Ò1Ø,9�L  Ð)à#×*Ñ*¨<Õ8ð9ð #Ð"ð VbÖ"bÐPQ A¨¯©«¡L´B·G±G¸A¿E¹E»GÀd¹NÓ4KÓ#KÐ"bÐÐ"bà"Ð"ùò #cs   Â:AC?Úone_waveformc           
      ó¸   — t        || j                  | j                  | j                  | j                  | j
                  | j                  d¬«      }|j                  S )zZ
        Extracts log-mel filterbank features for one waveform array (unbatched).
        Úlog10)r9   Úframe_lengthr   Ú
fft_lengthr:   r!   Úlog_mel)r   r9   r5   r6   r7   r:   r!   ÚT)r>   rV   Úlog_mel_specs      rA   Ú_extract_mel_featuresz.SpeechT5FeatureExtractor._extract_mel_features¡   sP   € ô #ØØ—;‘;Ø×)Ñ)Ø×)Ñ)Ø—z‘zØ×(Ñ(Ø—n‘nØô	
ˆð �~‰~ÐrB   ÚaudioÚaudio_targetÚpaddingÚ
max_lengthÚ
truncationÚpad_to_multiple_ofÚreturn_tensorsc
                 ó¶  — |€|€t        d«      ‚|	�;|	| j                  k7  rYt        d| › d| j                  › d| j                  › d|	› d�	«      ‚t        j                  d| j                  j
                  › d	�«       |� | j                  |d
||||||fi |
¤Ž}nd}|�> | j                  |d||||||fi |
¤Ž}|€|S |d   |d<   |j                  d«      }|�||d<   |S )aA  
        Main method to featurize and prepare for the model one or several sequence(s).

        Pass in a value for `audio` to extract waveform features. Pass in a value for `audio_target` to extract log-mel
        spectrogram features.

        Args:
            audio (`np.ndarray`, `List[float]`, `List[np.ndarray]`, `List[List[float]]`, *optional*):
                The sequence or batch of sequences to be processed. Each sequence can be a numpy array, a list of float
                values, a list of numpy arrays or a list of list of float values. This outputs waveform features. Must
                be mono channel audio, not stereo, i.e. single float per timestep.
            audio_target (`np.ndarray`, `List[float]`, `List[np.ndarray]`, `List[List[float]]`, *optional*):
                The sequence or batch of sequences to be processed as targets. Each sequence can be a numpy array, a
                list of float values, a list of numpy arrays or a list of list of float values. This outputs log-mel
                spectrogram features.
            padding (`bool`, `str` or [`~utils.PaddingStrategy`], *optional*, defaults to `False`):
                Select a strategy to pad the returned sequences (according to the model's padding side and padding
                index) among:

                - `True` or `'longest'`: Pad to the longest sequence in the batch (or no padding if only a single
                  sequence if provided).
                - `'max_length'`: Pad to a maximum length specified with the argument `max_length` or to the maximum
                  acceptable input length for the model if that argument is not provided.
                - `False` or `'do_not_pad'` (default): No padding (i.e., can output a batch with sequences of different
                  lengths).
            max_length (`int`, *optional*):
                Maximum length of the returned list and optionally padding length (see above).
            truncation (`bool`):
                Activates truncation to cut input sequences longer than *max_length* to *max_length*.
            pad_to_multiple_of (`int`, *optional*):
                If set will pad the sequence to a multiple of the provided value.

                This is especially useful to enable the use of Tensor Cores on NVIDIA hardware with compute capability
                `>= 7.5` (Volta), or on TPUs which benefit from having sequence lengths be a multiple of 128.
            return_attention_mask (`bool`, *optional*):
                Whether to return the attention mask. If left to the default, will return the attention mask according
                to the specific feature_extractor's default.

                [What are attention masks?](../glossary#attention-mask)

            return_tensors (`str` or [`~utils.TensorType`], *optional*):
                If set, will return tensors instead of list of python integers. Acceptable values are:

                - `'tf'`: Return TensorFlow `tf.constant` objects.
                - `'pt'`: Return PyTorch `torch.Tensor` objects.
                - `'np'`: Return Numpy `np.ndarray` objects.
            sampling_rate (`int`, *optional*):
                The sampling rate at which the `audio` or `audio_target` input was sampled. It is strongly recommended
                to pass `sampling_rate` at the forward call to prevent silent errors.
        Nz9You must provide either `audio` or `audio_target` values.z3The model corresponding to this feature extractor: z& was trained using a sampling rate of zB. Please make sure that the provided audio input was sampled with z	 and not ú.zDIt is strongly recommended to pass the `sampling_rate` argument to `zN()`. Failing to do so can result in silent errors that might be hard to debug.FTr   Úlabelsr   Údecoder_attention_mask)Ú
ValueErrorr   ÚloggerÚwarningr@   Ú__name__Ú_process_audioÚget)r>   r_   r`   ra   rb   rc   rd   r#   re   r   r?   ÚinputsÚinputs_targetri   s                 rA   Ú__call__z!SpeechT5FeatureExtractor.__call__´   se  € ð~ ˆ=˜\Ð1ÜÐXÓYÐYàÐ$Ø × 2Ñ 2Ò2Ü ØIÈ$Èð PØ×*Ñ*Ð+ð ,Ø×*Ñ*Ð+¨9°]°OÀ1ðFóð ô �N‰NØVÐW[×WeÑWe×WnÑWnÐVoð p\ð \ôð
 ÐØ(�T×(Ñ(ØØØØØØ"Ø%Øñ
ð ñ
‰Fð ˆFàÐ#Ø/˜D×/Ñ/ØØØØØØ"Ø%Øñ
ð ñ
ˆMð ˆ~Ø$Ð$à#0°Ñ#@��xÑ Ø)6×):Ñ):Ð;KÓ)LÐ&Ø)Ð5Ø7M�FÐ3Ñ4àˆrB   ÚspeechÚ	is_targetc	           	      óZ  — t        |t        j                  «      xr t        |j                  «      dkD  }
|
r&t        |j                  «      dkD  rt        d| › �«      ‚|
xs@ t        |t        t        f«      xr( t        |d   t        j                  t        t        f«      }|r4|D �cg c]'  }t        j                  |t        j                  ¬«      ‘Œ) c}}nª|s@t        |t        j                  «      s&t        j                  |t        j                  ¬«      }nht        |t        j                  «      rN|j                  t        j                  t        j                  «      u r|j                  t        j                  «      }|s|g}| j                  }|r=|D �cg c]  }| j                  |«      ‘Œ }}t        d|i«      }| j                   | _        nt        d|i«      } | j"                  |f|||||dœ|	¤Ž}|| _        |d   }t        |d   t        j                  «      s8|D �cg c]'  }t        j                  |t        j                  ¬«      ‘Œ) c}|d<   �nt        |t        j                  «      s€t        |d   t        j                  «      rc|d   j                  t        j                  t        j                  «      u r1|D �cg c]!  }|j                  t        j                  «      ‘Œ# c}|d<   nkt        |t        j                  «      rQ|j                  t        j                  t        j                  «      u r"|j                  t        j                  «      |d<   |j%                  d«      }|�6|D �cg c]'  }t        j                  |t        j&                  ¬«      ‘Œ) c}|d<   |sW| j(                  rK| j+                  ||¬	«      t,        j.                  ur|nd }| j1                  |d   || j2                  ¬
«      |d<   |�|j5                  |«      }|S c c}w c c}w c c}w c c}w c c}w )Nr&   r%   z2Only mono-channel audio is supported for input to r   )Údtyper   )ra   rb   rc   rd   r#   r   )rb   )r   r   )Ú
isinstancerF   ÚndarrayÚlenrN   rj   ÚlistÚtupleÚasarrayÚfloat32rv   Úfloat64Úastyper   r^   r   r   Úpadro   rH   r   Ú_get_padding_strategiesr   Ú
DO_NOT_PADrU   r   Úconvert_to_tensors)r>   rs   rt   ra   rb   rc   rd   r#   re   r?   Úis_batched_numpyÚ
is_batchedÚfeature_size_hackÚwaveformÚfeaturesÚencoded_inputsÚpadded_inputsr   rG   r   s                       rA   rn   z'SpeechT5FeatureExtractor._process_audio)  s|  € ô & f¬b¯j©jÓ9ÒS¼cÀ&Ç,Á,Ó>OÐRSÑ>SÐÙ¤ F§L¡LÓ 1°AÒ 5ÜÐQÐRVÐQWÐXÓYÐYØ%ò 
Ü�v¤¤e˜}Ó-Òd´:¸fÀQ¹iÌ"Ï*É*ÔV[Ô]aÐIbÓ3cð 	ñ ØIOÖP¸v”b—j‘j ¬r¯z©zÖ:ÒP‰FÙ¤J¨v´r·z±zÔ$BÜ—Z‘Z ¬b¯j©jÔ9‰FÜ˜¤§
¡
Ô+°·±ÄÇÁÌÏÉÓ@TÑ0TØ—]‘]¤2§:¡:Ó.ˆFñ Ø�XˆFð !×-Ñ-Ðñ ØMSÖTÀ˜×2Ñ2°8Õ<ÐTˆHÐTÜ)¨>¸8Ð*DÓEˆNØ $× 1Ñ 1ˆDÕä)¨>¸6Ð*BÓCˆNà ˜Ÿ™Øð
àØ!Ø!Ø1Ø"7ñ
ð ñ
ˆð .ˆÔð % ^Ñ4ˆÜ˜, q™/¬2¯:©:Ô6Ø^jÖ,kÐUZ¬R¯Z©Z¸ÄRÇZÁZÖ-PÒ,kˆM˜.Ó)ä˜<¬¯©Ô4Ü˜<¨™?¬B¯J©JÔ7Ø˜Q‘×%Ñ%¬¯©´"·*±*Ó)=Ñ=àS_Ö,`È%¨U¯\©\¼"¿*¹*Õ-EÒ,`ˆM˜.Ò)Ü˜¤b§j¡jÔ1°l×6HÑ6HÌBÏHÉHÔUW×U_ÑU_ÓL`Ñ6`Ø,8×,?Ñ,?ÄÇ
Á
Ó,KˆM˜.Ñ)ð '×*Ñ*Ð+;Ó<ˆØÐ%Ø^lÖ.mÐUZ¬r¯z©z¸%ÄrÇxÁxÖ/PÒ.mˆMÐ*Ñ+ñ ˜T×.Ò.ð ×/Ñ/°ÀJÐ/ÓOÔWf×WqÑWqÑqñ àð ð
 -1×,HÑ,HØ˜nÑ-¸nÐ\`×\nÑ\nð -Ió -ˆM˜.Ñ)ð Ð%Ø)×<Ñ<¸^ÓLˆMàÐùòC Qùò Uùò* -lùò -aùò /ns   Â',PÆPÈ$,PË&P#Í4,P(c                 óJ   •— t         ‰| �  «       }g d¢}|D ]
  }||v sŒ||= Œ |S )N)r9   r:   r5   r6   r7   r8   )r3   Úto_dict)r>   ÚoutputÚnamesr(   r@   s       €rA   rŒ   z SpeechT5FeatureExtractor.to_dict€  s;   ø€ Ü‘‘Ó"ˆò ^ˆØò 	!ˆDØ�vŠ~Ø˜4‘Lð	!ð ˆrB   )r&   i€>  ç        FéP   é   é@   Úhann_windowr1   r�   i°  g»½×Ùß|Û=r%   T)r�   )	NNFNFNNNN)FFNFNNN)rm   Ú
__module__Ú__qualname__Ú__doc__Úmodel_input_namesÚintÚfloatÚboolÚstrr4   Ústaticmethodr   rF   rx   rU   r^   r   r   r   r   r   rr   rn   r   r   rŒ   Ú__classcell__)r@   s   @rA   r   r      sI  ø„ ñ*ðX (Ð)9Ð:Ðð Ø"Ø"Ø"ØØØØ)Ø$'ØØØ Ø !Ø&*ñ:àð:ð ð:ð ð	:ð
 ð:ð ð:ð ð:ð ð:ð ð:ð "ð:ð ð:ð ð:ð ð:ð ð:ð  $õ:ðx ð beñ#Ø˜2Ÿ:™:Ñ&ð#Ø8<¸R¿Z¹ZÑ8Hð#ØY^ð#à	ˆb�j‰jÑ	ò#ó ð#ð*à—j‘jðð 
�‰óð* `dØfjØ5:Ø$(Ø Ø,0Ø04Ø;?Ø'+ñsà˜˜bŸj™j¨$¨u©+°t¸B¿J¹JÑ7GÈÈdÐSXÉkÑIZÐZÑ[Ñ\ðsð ˜u R§Z¡Z°°e±¸dÀ2Ç:Á:Ñ>NÐPTÐUYÐZ_ÑU`ÑPaÐ%aÑbÑcðsð �t˜S /Ð1Ñ2ð	sð
 ˜S‘Mðsð ðsð % S™Mðsð  (¨™~ðsð !  s¨J Ñ!7Ñ8ðsð   ‘}ðsð 
ósðp  Ø5:Ø$(Ø Ø,0Ø04Ø;?ñUà�b—j‘j $ u¡+¨t°B·J±JÑ/?ÀÀdÈ5ÁkÑARÐRÑSðUð ðUð �t˜S /Ð1Ñ2ð	Uð
 ˜S‘MðUð ðUð % S™MðUð  (¨™~ðUð !  s¨J Ñ!7Ñ8ðUð 
óUðn	˜˜c 3˜h™÷ 	ñ 	rB   r   )r–   r;   Útypingr   r   r   r   r   ÚnumpyrF   Úaudio_utilsr	   r
   r   r   Ú!feature_extraction_sequence_utilsr   Úfeature_extraction_utilsr   Úutilsr   r   r   Ú
get_loggerrm   rk   r   Ú__all__r2   rB   rA   ú<module>r¦      sV   ðñ ,ã ß 3Õ 3ã ç \Ó \Ý IÝ 4ß 9Ñ 9ð 
ˆ×	Ñ	˜HÓ	%€ôjÐ7ô jðZ &Ð
&�rB   