Ë
    T^(h.  ã                   óÀ   — d dl Z d dlmZmZ d dlmZ ddlmZ erddlm	Z	 ddl
mZmZmZmZmZ dd	lmZmZ  e«       rd dlZ ej(                  e«      Z G d
„ de«      Zy)é    N)ÚTYPE_CHECKINGÚOptional)Úversioné   )ÚHfQuantizeré   )ÚPreTrainedModel)Úis_auto_gptq_availableÚis_gptqmodel_availableÚis_optimum_availableÚis_torch_availableÚlogging)Ú
GPTQConfigÚQuantizationConfigMixinc                   ó‚   ‡ — e Zd ZdZdZg d¢ZdZdefˆ fd„Zd„ Z	dd„Z
d	„ Zdd„Zdd„Zedd
ed   fd„«       Zdd„Zˆ xZS )ÚGptqHfQuantizerzå
    Quantizer of the GPTQ method - for GPTQ the quantizer support calibration of the model through
    `auto_gptq` or `gptqmodel` package. Quantization is done under the hood for users if they load a non-prequantized model.
    F)ÚoptimumÚ	auto_gptqÚ	gptqmodelNÚquantization_configc                 ó¸   •— t        ‰| �  |fi |¤Ž t        «       st        d«      ‚ddlm} |j                  | j                  j                  «       «      | _	        y )NúGLoading a GPTQ quantized model requires optimum (`pip install optimum`)r   )ÚGPTQQuantizer)
ÚsuperÚ__init__r   ÚImportErrorÚoptimum.gptqr   Ú	from_dictr   Úto_dict_optimumÚoptimum_quantizer)Úselfr   Úkwargsr   Ú	__class__s       €úd/var/www/skyplay_api_hub/venv/lib/python3.12/site-packages/transformers/quantizers/quantizer_gptq.pyr   zGptqHfQuantizer.__init__-   sM   ø€ Ü‰ÑÐ,Ñ7°Ò7ä#Ô%ÜÐgÓhÐhÝ.à!.×!8Ñ!8¸×9QÑ9Q×9aÑ9aÓ9cÓ!dˆÕó    c                 óÚ  — t        «       st        d«      ‚t        «       rt        «       rt        j                  d«       t        «       xrH t        j                  t        j                  j                  d«      «      t        j                  d«      kD  xs
 t        «       }|s)t        j                  j                  «       st        d«      ‚t        «       st        «       st        d«      ‚t        «       rSt        j                  t        j                  j                  d«      «      t        j                  d«      k  rt        d«      ‚t        «       rœt        j                  t        j                  j                  d	«      «      t        j                  d
«      k  sHt        j                  t        j                  j                  d«      «      t        j                  d«      k  rt        d«      ‚y y )Nr   z4Detected gptqmodel and auto-gptq, will use gptqmodelz	auto-gptqz0.4.2z2GPU is required to quantize or run quantize model.z|Loading a GPTQ quantized model requires gptqmodel (`pip install gptqmodel`) or auto-gptq (`pip install auto-gptq`) library. r   z‹You need a version of auto_gptq >= 0.4.2 to use GPTQ: `pip install --upgrade auto-gptq` or use gptqmodel by `pip install gptqmodel>=1.4.3`.r   z1.4.3r   ú1.23.99zJThe gptqmodel version should be >= 1.4.3, optimum version should >= 1.24.0)r   r   r
   r   ÚloggerÚwarningr   ÚparseÚ	importlibÚmetadataÚtorchÚcudaÚis_availableÚRuntimeError)r!   Úargsr"   Úgptq_supports_cpus       r$   Úvalidate_environmentz$GptqHfQuantizer.validate_environment6   s€  € Ü#Ô%ÜÐgÓhÐhÜ!Ô#Ô(>Ô(@Ü�N‰NÐQÔRô #Ó$ò `Ü—‘œi×0Ñ0×8Ñ8¸ÓEÓFÌÏÉÐW^ÓI_Ñ_ò&ô $Ó%ð 	ñ !¬¯©×)@Ñ)@Ô)BÜÐSÓTÐTÜ(Ô*Ô.DÔ.FÜð Oóð ô $Ô%¬'¯-©-¼	×8JÑ8J×8RÑ8RÐS^Ó8_Ó*`Ôcj×cpÑcpØód
ò +
ô ð ^óð ô $Ô%Ü�M‰Mœ)×,Ñ,×4Ñ4°[ÓAÓBÄWÇ]Á]ÐSZÓE[Ò[Ü�}‰}œY×/Ñ/×7Ñ7¸	ÓBÓCÄgÇmÁmÐT]ÓF^Ò^äÐjÓkÐkð _ð &r%   c                 ó¨   — |€'t         j                  }t        j                  d«       |S |t         j                  k7  rt        j                  d«       |S )NzRLoading the model in `torch.float16`. To overwrite it, set `torch_dtype` manually.zRWe suggest you to set `torch_dtype=torch.float16` for better efficiency with GPTQ.)r-   Úfloat16r(   Úinfo)r!   Útorch_dtypes     r$   Úupdate_torch_dtypez"GptqHfQuantizer.update_torch_dtypeR   sG   € ØÐÜŸ-™-ˆKÜ�K‰KÐlÔmð Ðð œEŸM™MÒ)Ü�K‰KÐlÔmØÐr%   c                 ó�   — |€dt        j                  d«      i}t        «       s"|ddt        j                  d«      ifv r|ddik(   |S )NÚ Úcpur   )r-   Údevicer   )r!   Ú
device_maps     r$   Úupdate_device_mapz!GptqHfQuantizer.update_device_mapZ   sN   € ØÐØœeŸl™l¨5Ó1Ð2ˆJä%Ô'¨J¸5À2ÄuÇ|Á|ÐTYÓGZÐB[Ð:\Ñ,\Ø˜2˜q˜'Ò!ØÐr%   Úmodelr	   c                 óh  — |j                   j                  dk7  rt        d«      ‚| j                  r‚t	        j
                  t        j                  j	                  d«      «      t	        j
                  d«      k  r| j                  j                  |«      }y  | j                  j                  |fi |¤Ž}y y )NÚ	input_idsz%We can only quantize pure text model.r   r'   )
r#   Úmain_input_namer0   Úpre_quantizedr   r*   r+   r,   r    Úconvert_model©r!   r?   r"   s      r$   Ú$_process_model_before_weight_loadingz4GptqHfQuantizer._process_model_before_weight_loadingb   s�   € Ø�?‰?×*Ñ*¨kÒ9ÜÐFÓGÐGà×Òä�}‰}œY×/Ñ/×7Ñ7¸	ÓBÓCÄwÇ}Á}ÐU^ÓG_Ò_Ø×.Ñ.×<Ñ<¸UÓC‘à<˜×.Ñ.×<Ñ<¸UÑMÀfÑM‘ð r%   c                 óŽ  — | j                   r| j                  j                  |«      }y | j                  j                  €|j
                  | j                  _        | j                  j                  || j                  j                  «       t        j                  | j                  j                  «       «      |j                  _        y ©N)rC   r    Úpost_init_modelr   Ú	tokenizerÚname_or_pathÚquantize_modelr   r   Úto_dictÚconfigrE   s      r$   Ú#_process_model_after_weight_loadingz3GptqHfQuantizer._process_model_after_weight_loadingm   s�   € Ø×ÒØ×*Ñ*×:Ñ:¸5ÓA‰Eà×'Ñ'×1Ñ1Ð9Ø5:×5GÑ5G�×(Ñ(Ô2à×"Ñ"×1Ñ1°%¸×9QÑ9Q×9[Ñ9[Ô\Ü/9×/CÑ/CÀD×DZÑDZ×DbÑDbÓDdÓ/eˆE�L‰LÕ,r%   c                  ó   — y©NT© )r!   r?   s     r$   Úis_trainablezGptqHfQuantizer.is_trainablew   s   € àr%   c                  ó   — yrQ   rR   )r!   Úsafe_serializations     r$   Úis_serializablezGptqHfQuantizer.is_serializable{   s   € Ør%   )r7   útorch.dtypeÚreturnrW   )r?   r	   rH   )Ú__name__Ú
__module__Ú__qualname__Ú__doc__Úrequires_calibrationÚrequired_packagesr    r   r   r3   r8   r>   rF   rO   Úpropertyr   rS   rV   Ú__classcell__)r#   s   @r$   r   r   #   sm   ø„ ñð
 !ÐÚ=ÐØÐðeÐ,Cõ eòló8òó	Nófð ñ (Ð+<Ñ"=ò ó ð÷r%   r   )r+   Útypingr   r   Ú	packagingr   Úbaser   Úmodeling_utilsr	   Úutilsr
   r   r   r   r   Úutils.quantization_configr   r   r-   Ú
get_loggerrY   r(   r   rR   r%   r$   ú<module>rh      sO   ðó ß *å å ñ Ý0ç uÕ uß Kñ ÔÛà	ˆ×	Ñ	˜HÓ	%€ôY�kõ Yr%   