Ë
    Hêñi—  ã                   óÄ   — d dl mZ d dlmZ d dlmZ ddlmZ erddlm	Z	 ddl
mZmZmZmZ dd	lmZmZ  e«       rd d
lZ ej&                  e«      ZdZdZ G d„ de«      Zy
)é    )Úmetadata)ÚTYPE_CHECKING)Úversioné   )ÚHfQuantizeré   )ÚPreTrainedModel)Úis_gptqmodel_availableÚis_optimum_availableÚis_torch_availableÚlogging)Ú
GPTQConfigÚQuantizationConfigMixinNz1.4.3z1.24.0c                   óx   ‡ — e Zd ZU dZdZded<   defˆ fd„Zd„ Zdd„Z	d	„ Z
dd
„Zdd„Zedefd„«       Zd„ Zˆ xZS )ÚGptqHfQuantizerzþ
    Quantizer of the GPTQ method - for GPTQ the quantizer support calibration of the model through
    the GPT-QModel package (Python import name `gptqmodel`). Quantization is done under the hood for users if they
    load a non-prequantized model.
    Fr   Úquantization_configc                 ó¸   •— t        ‰| �  |fi |¤Ž t        «       st        d«      ‚ddlm} |j                  | j                  j                  «       «      | _	        y )NúGLoading a GPTQ quantized model requires optimum (`pip install optimum`)r   )ÚGPTQQuantizer)
ÚsuperÚ__init__r   ÚImportErrorÚoptimum.gptqr   Ú	from_dictr   Úto_dict_optimumÚoptimum_quantizer)Úselfr   Úkwargsr   Ú	__class__s       €úh/var/www/pod-logistic/pod-ai/venv/lib/python3.12/site-packages/transformers/quantizers/quantizer_gptq.pyr   zGptqHfQuantizer.__init__1   sM   ø€ Ü‰ÑÐ,Ñ7°Ò7ä#Ô%ÜÐgÓhÐhÝ.à!.×!8Ñ!8¸×9QÑ9Q×9aÑ9aÓ9cÓ!dˆÕó    c                 ó  — t        «       st        d«      ‚t        «       }|s)t        j                  j                  «       st        d«      ‚t        «       st        d«      ‚t        «       ržt        j                  t        j                  d«      «      t        j                  t        «      k  sBt        j                  t        j                  d«      «      t        j                  t        «      k  rt        dt        › dt        › �«      ‚y y )Nr   z2GPU is required to quantize or run quantize model.zTLoading a GPTQ quantized model requires gptqmodel (`pip install gptqmodel`) library.Ú	gptqmodelÚoptimumz#The gptqmodel version should be >= z, optimum version should >= )r   r   r
   ÚtorchÚcudaÚis_availableÚRuntimeErrorr   Úparser   ÚMIN_GPTQ_VERSIONÚMIN_OPTIMUM_VERSION)r   Úargsr   Úgptq_supports_cpus       r    Úvalidate_environmentz$GptqHfQuantizer.validate_environment:   sÌ   € Ü#Ô%ÜÐgÓhÐhä2Ó4ÐÙ ¬¯©×)@Ñ)@Ô)BÜÐSÓTÐTÜ'Ô)ÜÐtÓuÐuÜ#Ô%Ü�M‰Mœ(×*Ñ*¨;Ó7Ó8¼7¿=¹=ÔIYÓ;ZÒZÜ�}‰}œX×-Ñ-¨iÓ8Ó9¼G¿M¹MÔJ]Ó<^Ò^äØ5Ô6FÐ5GÐGcÔdwÐcxÐyóð ð _ð &r!   Úreturnc                 óV   — |t         j                  k7  rt        j                  d«       |S )NzLWe suggest you to set `dtype=torch.float16` for better efficiency with GPTQ.)r%   Úfloat16ÚloggerÚinfo)r   Údtypes     r    Úupdate_dtypezGptqHfQuantizer.update_dtypeK   s    € Ø”E—M‘MÒ!Ü�K‰KÐfÔgØˆr!   c                 ó8   — |€dt        j                  d«      i}|S )NÚ Úcpu)r%   Údevice)r   Ú
device_maps     r    Úupdate_device_mapz!GptqHfQuantizer.update_device_mapP   s!   € ØÐØœeŸl™l¨5Ó1Ð2ˆJØÐr!   c                 ó\  — |j                   j                  dk7  rt        d«      ‚| j                  r|t	        j
                  t        j                  d«      «      t	        j
                  t        «      k  r| j                  j                  |«      }y  | j                  j                  |fi |¤Ž}y y )NÚ	input_idsz%We can only quantize pure text model.r$   )
r   Úmain_input_namer(   Úpre_quantizedr   r)   r   r+   r   Úconvert_model©r   Úmodelr   s      r    Ú$_process_model_before_weight_loadingz4GptqHfQuantizer._process_model_before_weight_loadingU   s‡   € Ø�?‰?×*Ñ*¨kÒ9ÜÐFÓGÐGà×Òä�}‰}œX×-Ñ-¨iÓ8Ó9¼G¿M¹MÔJ]Ó<^Ò^Ø×.Ñ.×<Ñ<¸UÓC‘à<˜×.Ñ.×<Ñ<¸UÑMÀfÑM‘ð r!   c                 óŽ  — | j                   r| j                  j                  |«      }y | j                  j                  €|j
                  | j                  _        | j                  j                  || j                  j                  «       t        j                  | j                  j                  «       «      |j                  _        y )N)r?   r   Úpost_init_modelr   Ú	tokenizerÚname_or_pathÚquantize_modelr   r   Úto_dictÚconfigrA   s      r    Ú#_process_model_after_weight_loadingz3GptqHfQuantizer._process_model_after_weight_loading`   s�   € Ø×ÒØ×*Ñ*×:Ñ:¸5ÓA‰Eà×'Ñ'×1Ñ1Ð9Ø5:×5GÑ5G�×(Ñ(Ô2à×"Ñ"×1Ñ1°%¸×9QÑ9Q×9[Ñ9[Ô\Ü/9×/CÑ/CÀD×DZÑDZ×DbÑDbÓDdÓ/eˆE�L‰LÕ,r!   c                  ó   — y©NT© ©r   s    r    Úis_trainablezGptqHfQuantizer.is_trainablej   s   € àr!   c                  ó   — yrM   rN   rO   s    r    Úis_serializablezGptqHfQuantizer.is_serializablen   s   € Ør!   )r4   útorch.dtyper/   rS   )rB   r	   )Ú__name__Ú
__module__Ú__qualname__Ú__doc__Úrequires_calibrationÚ__annotations__r   r   r.   r5   r;   rC   rK   ÚpropertyÚboolrP   rR   Ú__classcell__)r   s   @r    r   r   '   s`   ø… ñð !ÐØ%Ó%ðeÐ,Cõ eòó"ò
ó
	Nófð ð˜dò ó ðör!   r   )Ú	importlibr   Útypingr   Ú	packagingr   Úbaser   Úmodeling_utilsr	   Úutilsr
   r   r   r   Úutils.quantization_configr   r   r%   Ú
get_loggerrT   r2   r*   r+   r   rN   r!   r    ú<module>re      s]   ðõ Ý  å å ñ Ý0ç ]Ó ]ß Kñ ÔÛà	ˆ×	Ñ	˜HÓ	%€ð Ð ØÐ ôH�kõ Hr!   