Ë
    HêñiÁ  ã                   ó¬   — d dl mZ ddlmZ ddlmZ erddlmZ ddlm	Z	m
Z
mZmZ ddlmZ  e«       rd d	lZ ej                   e«      Z G d
„ de«      Zy	)é    )ÚTYPE_CHECKINGé   )ÚHfQuantizer)Úget_module_from_nameé   )ÚPreTrainedModel)Úis_accelerate_availableÚis_optimum_quanto_availableÚis_torch_availableÚlogging)ÚQuantoConfigNc                   óÈ   ‡ — e Zd ZU dZdZded<   defˆ fd„Zd„ Zddd	e	d
e
fd„Zdee	ee	z  f   d
ee	ee	z  f   fd„Zddd	e	ddd
efˆ fd„Zdd„Zed
e
fd„«       Zd„ Zd„ Zˆ xZS )ÚQuantoHfQuantizerz*
    Quantizer for the quanto library
    Fr   Úquantization_configc                 óŠ   •— t        ‰| �  |fi |¤Ž dddddœ}|j                  | j                  j                  d «      | _        y )Nr   g      à?g      Ð?)Úint8Úfloat8Úint4Úint2)ÚsuperÚ__init__Úgetr   ÚweightsÚquantized_param_size)Úselfr   ÚkwargsÚmap_to_param_sizeÚ	__class__s       €új/var/www/pod-logistic/pod-ai/venv/lib/python3.12/site-packages/transformers/quantizers/quantizer_quanto.pyr   zQuantoHfQuantizer.__init__.   sN   ø€ Ü‰ÑÐ,Ñ7°Ò7àØØØñ	
Ðð %6×$9Ñ$9¸$×:RÑ:R×:ZÑ:ZÐ\`Ó$aˆÕ!ó    c                 óV  — t        «       st        d«      ‚t        «       st        d«      ‚|j                  d«      }t	        |t
        «      r=t        |«      dkD  rd|j                  «       v sd|j                  «       v rt        d«      ‚| j                  j                  �t        d«      ‚y )	NzhLoading an optimum-quanto quantized model requires optimum-quanto library (`pip install optimum-quanto`)z`Loading an optimum-quanto quantized model requires accelerate library (`pip install accelerate`)Ú
device_mapr   ÚcpuÚdiskzÜYou are attempting to load an model with a device_map that contains a CPU or disk device.This is not supported with quanto when the model is quantized on the fly. Please remove the CPU or disk device from the device_map.zÂWe don't support quantizing the activations with transformers library.Use quanto library for more complex use cases such as activations quantization, calibration and quantization aware training.)r
   ÚImportErrorr	   r   Ú
isinstanceÚdictÚlenÚvaluesÚ
ValueErrorr   Úactivations)r   Úargsr   r"   s       r   Úvalidate_environmentz&QuantoHfQuantizer.validate_environment8   s¶   € Ü*Ô,ÜØzóð ô 'Ô(ÜØróð ð —Z‘Z Ó-ˆ
Ü�j¤$Ô'Ü�:‹ Ò" u°
×0AÑ0AÓ0CÑ'CÀvÐQ[×QbÑQbÓQdÑGdÜ ðPóð ð
 ×#Ñ#×/Ñ/Ð;ÜðOóð ð <r    Úmodelr   Ú
param_nameÚreturnc                 óh   — ddl m} t        ||«      \  }}t        ||«      rd|v r|j                   S y)Nr   )ÚQModuleMixinÚweightF)Úoptimum.quantor2   r   r&   Úfrozen)r   r.   r/   r   r2   ÚmoduleÚtensor_names          r   Úparam_needs_quantizationz*QuantoHfQuantizer.param_needs_quantizationO   s7   € Ý/ä2°5¸*ÓEÑˆ�ä�f˜lÔ+°¸KÑ0Gà—}‘}Ð$Ð$àr    Ú
max_memoryc                 ó^   — |j                  «       D ��ci c]  \  }}||dz  “Œ }}}|S c c}}w )NgÍÌÌÌÌÌì?)Úitems)r   r9   ÚkeyÚvals       r   Úadjust_max_memoryz#QuantoHfQuantizer.adjust_max_memoryZ   s6   € Ø6@×6FÑ6FÓ6H×I©(¨#¨s�c˜3 ™:‘oÐIˆ
ÑIØÐùó Js   ”)Úparamztorch.Tensorc                 óz   •— | j                  ||«      r| j                  �| j                  S t        ‰| �  |||«      S )z4Return the element size (in bytes) for `param_name`.)r8   r   r   Úparam_element_size)r   r.   r/   r?   r   s       €r   rA   z$QuantoHfQuantizer.param_element_size^   s>   ø€ à×(Ñ(¨°
Ô;À×@YÑ@YÐ@eØ×,Ñ,Ð,ä‰wÑ)¨%°¸UÓCÐCr    c                 óº   — ddl m} | j                  || j                  j                  |j
                  «      | _         ||| j                  | j                  ¬«      }y )Nr   )Úreplace_with_quanto_layers)Úmodules_to_not_convertr   )ÚintegrationsrC   Úget_modules_to_not_convertr   rD   Ú_keep_in_fp32_modules)r   r.   r   rC   s       r   Ú$_process_model_before_weight_loadingz6QuantoHfQuantizer._process_model_before_weight_loadinge   sQ   € Ý=à&*×&EÑ&EØ�4×+Ñ+×BÑBÀE×D_ÑD_ó'
ˆÔ#ñ +Ø¨$×*EÑ*EÐ[_×[sÑ[sô
‰r    c                  ó   — y)NT© ©r   s    r   Úis_trainablezQuantoHfQuantizer.is_trainablep   s   € àr    c                  ó   — y)NFrJ   rK   s    r   Úis_serializablez!QuantoHfQuantizer.is_serializablet   s   € Ør    c                 ó   — ddl m}  || «      S )Nr   )ÚQuantoQuantize)Úintegrations.quantorP   )r   rP   s     r   Úget_quantize_opsz"QuantoHfQuantizer.get_quantize_opsw   s   € Ý8á˜dÓ#Ð#r    )r.   r   )Ú__name__Ú
__module__Ú__qualname__Ú__doc__Úrequires_calibrationÚ__annotations__r   r   r-   ÚstrÚboolr8   r'   Úintr>   ÚfloatrA   rH   ÚpropertyrL   rN   rR   Ú__classcell__)r   s   @r   r   r   &   sÆ   ø… ñð !ÐØ'Ó'ðb¨Lõ bòð.	Ð.?ð 	ÈSð 	Ð_có 	ð¨D°°c¸C±i°Ñ,@ð ÀTÈ#ÈsÐUXÉyÈ.ÑEYó ðDÐ(9ð DÀsð DÐSað DÐfkõ Dó	
ð ð˜dò ó ðòö$r    r   )Útypingr   Úbaser   Úquantizers_utilsr   Úmodeling_utilsr   Úutilsr	   r
   r   r   Úutils.quantization_configr   ÚtorchÚ
get_loggerrS   Úloggerr   rJ   r    r   ú<module>rh      sR   ðõ !å Ý 2ñ Ý0÷ó õ 5ñ ÔÛà	ˆ×	Ñ	˜HÓ	%€ôT$˜õ T$r    