ó
    qyüi—  ã                   óØ   • S SK Jr  S SKJr  S SKJr  SSKJr  \(       a  SSKJ	r	  SSK
JrJrJrJr  SS	KJrJr  \" 5       (       a  S S
Kr\R&                  " \5      rSrSr " S S\5      rg
)é    )Úmetadata)ÚTYPE_CHECKING)Úversioné   )ÚHfQuantizeré   )ÚPreTrainedModel)Úis_gptqmodel_availableÚis_optimum_availableÚis_torch_availableÚlogging)Ú
GPTQConfigÚQuantizationConfigMixinNz1.4.3z1.24.0c                   óŒ   ^ • \ rS rSr% SrSrS\S'   S\4U 4S jjrS r	SS	 jr
S
 rSS jrSS jr\S\4S j5       rS rSrU =r$ )ÚGptqHfQuantizeré'   zî
Quantizer of the GPTQ method - for GPTQ the quantizer support calibration of the model through
the GPT-QModel package (Python import name `gptqmodel`). Quantization is done under the hood for users if they
load a non-prequantized model.
Fr   Úquantization_configc                 óÄ   >• [         TU ]  " U40 UD6  [        5       (       d  [        S5      eSSKJn  UR                  U R                  R                  5       5      U l	        g )NúGLoading a GPTQ quantized model requires optimum (`pip install optimum`)r   )ÚGPTQQuantizer)
ÚsuperÚ__init__r   ÚImportErrorÚoptimum.gptqr   Ú	from_dictr   Úto_dict_optimumÚoptimum_quantizer)Úselfr   Úkwargsr   Ú	__class__s       €Úc/home/mande/repo/quber/.venv/lib/python3.13/site-packages/transformers/quantizers/quantizer_gptq.pyr   ÚGptqHfQuantizer.__init__1   sP   ø€ Ü‰ÒÐ,Ñ7°Ò7ä#×%Ñ%ÜÐgÓhÐhÝ.à!.×!8Ñ!8¸×9QÑ9Q×9aÑ9aÓ9cÓ!dˆÕó    c                 óT  • [        5       (       d  [        S5      e[        5       nU(       d.  [        R                  R                  5       (       d  [        S5      e[        5       (       d  [        S5      e[        5       (       a¦  [        R                  " [        R                  " S5      5      [        R                  " [        5      :  dF  [        R                  " [        R                  " S5      5      [        R                  " [        5      :  a  [        S[         S[         35      eg g )Nr   z2GPU is required to quantize or run quantize model.zTLoading a GPTQ quantized model requires gptqmodel (`pip install gptqmodel`) library.Ú	gptqmodelÚoptimumz#The gptqmodel version should be >= z, optimum version should >= )r   r   r
   ÚtorchÚcudaÚis_availableÚRuntimeErrorr   Úparser   ÚMIN_GPTQ_VERSIONÚMIN_OPTIMUM_VERSION)r   Úargsr   Úgptq_supports_cpus       r!   Úvalidate_environmentÚ$GptqHfQuantizer.validate_environment:   sØ   € Ü#×%Ñ%ÜÐgÓhÐhä2Ó4ÐÞ ¬¯©×)@Ñ)@×)BÑ)BÜÐSÓTÐTÜ'×)Ñ)ÜÐtÓuÐuÜ#×%Ñ%Ü�MŠMœ(×*Ò*¨;Ó7Ó8¼7¿=º=ÔIYÓ;ZÓZÜ�}Š}œX×-Ò-¨iÓ8Ó9¼G¿MºMÔJ]Ó<^Ó^äØ5Ô6FÐ5GÐGcÔdwÐcxÐyóð ð _ð &r#   Úreturnc                 óX   • U[         R                  :w  a  [        R                  S5        U$ )NzLWe suggest you to set `dtype=torch.float16` for better efficiency with GPTQ.)r'   Úfloat16ÚloggerÚinfo)r   Údtypes     r!   Úupdate_dtypeÚGptqHfQuantizer.update_dtypeK   s    € Ø”E—M‘MÓ!Ü�K‰KÐfÔgØˆr#   c                 ó<   • Uc  S[         R                  " S5      0nU$ )NÚ Úcpu)r'   Údevice)r   Ú
device_maps     r!   Úupdate_device_mapÚ!GptqHfQuantizer.update_device_mapP   s!   € ØÑØœeŸlšl¨5Ó1Ð2ˆJØÐr#   c                 óp  • UR                   R                  S:w  a  [        S5      eU R                  (       a€  [        R
                  " [        R                  " S5      5      [        R
                  " [        5      :  a  U R                  R                  U5      ng U R                  R                  " U40 UD6ng g )NÚ	input_idsz%We can only quantize pure text model.r&   )
r    Úmain_input_namer*   Úpre_quantizedr   r+   r   r-   r   Úconvert_model©r   Úmodelr   s      r!   Ú$_process_model_before_weight_loadingÚ4GptqHfQuantizer._process_model_before_weight_loadingU   s…   € Ø�?‰?×*Ñ*¨kÓ9ÜÐFÓGÐGà××ä�}Š}œX×-Ò-¨iÓ8Ó9¼G¿MºMÔJ]Ó<^Ó^Ø×.Ñ.×<Ñ<¸UÓC‘à×.Ñ.×<Ò<¸UÑMÀfÑM‘ð r#   c                 óš  • U R                   (       a  U R                  R                  U5      ng U R                  R                  c  UR
                  U R                  l        U R                  R                  XR                  R                  5        [        R                  " U R                  R                  5       5      UR                  l        g )N)rD   r   Úpost_init_modelr   Ú	tokenizerÚname_or_pathÚquantize_modelr   r   Úto_dictÚconfigrF   s      r!   Ú#_process_model_after_weight_loadingÚ3GptqHfQuantizer._process_model_after_weight_loading`   s�   € Ø××Ø×*Ñ*×:Ñ:¸5ÓA‰Eà×'Ñ'×1Ñ1Ñ9Ø5:×5GÑ5G�×(Ñ(Ô2à×"Ñ"×1Ñ1°%×9QÑ9Q×9[Ñ9[Ô\Ü/9×/CÒ/CÀD×DZÑDZ×DbÑDbÓDdÓ/eˆE�L‰LÕ,r#   c                 ó   • g©NT© ©r   s    r!   Úis_trainableÚGptqHfQuantizer.is_trainablej   s   € àr#   c                 ó   • grT   rU   rV   s    r!   Úis_serializableÚGptqHfQuantizer.is_serializablen   s   € Ør#   )r   )r7   útorch.dtyper2   r\   )rG   r	   )Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__Úrequires_calibrationÚ__annotations__r   r   r0   r8   r?   rH   rQ   ÚpropertyÚboolrW   rZ   Ú__static_attributes__Ú__classcell__)r    s   @r!   r   r   '   se   ø‡ ñð !ÐØ%Ó%ðeÐ,C÷ eòô"ò
ô
	Nôfð ð˜dó ó ð÷ð r#   r   )Ú	importlibr   Útypingr   Ú	packagingr   Úbaser   Úmodeling_utilsr	   Úutilsr
   r   r   r   Úutils.quantization_configr   r   r'   Ú
get_loggerr]   r5   r,   r-   r   rU   r#   r!   Ú<module>rp      s^   ðõ Ý  å å ö Ý0ç ]Ó ]ß Kñ ×ÑÛà	×	Ò	˜HÓ	%€ð Ð ØÐ ôH�kõ Hr#   