ó
    qyüi^  ã                   óÀ   • S SK Jr  SSKJr  \(       a  SSKJr  SSKJr  SSKJ	r	J
r
JrJr  SSKJr  \" 5       (       a  S S	Kr\R                   " \5      r " S
 S\5      rg	)é    )ÚTYPE_CHECKINGé   )ÚHfQuantizeré   )ÚPreTrainedModel)Ú
EetqConfig)Úis_accelerate_availableÚis_kernels_availableÚis_torch_availableÚlogging)Úget_module_from_nameNc                   ó”   ^ • \ rS rSr% SrSrS\S'   U 4S jrS rSS	 jr	S
SS\
S\4S jr  SS jrS r\S\4S j5       rS rSrU =r$ )ÚEetqHfQuantizeré"   z2
8-bit quantization from EETQ quantization method
Fr   Úquantization_configc                 ó(   >• [         TU ]  " U40 UD6  g )N)ÚsuperÚ__init__)Úselfr   ÚkwargsÚ	__class__s      €Úc/home/mande/repo/quber/.venv/lib/python3.13/site-packages/transformers/quantizers/quantizer_eetq.pyr   ÚEetqHfQuantizer.__init__*   s   ø€ Ü‰ÒÐ,Ñ7°Ó7ó    c                 óÌ  • [        5       (       d  [        S5      e[        5       (       d  [        S5      e[        R                  R                  5       (       d  [        S5      eUR                  S5      nUc  [        R                  S5        g [        U[        5      (       aC  [        U5      S:”  a  SUR                  5       ;   d  SUR                  5       ;   a  [        S	5      eg g )
NzHLoading an EETQ quantized model requires kernels (`pip install kernels`)zNLoading an EETQ quantized model requires accelerate (`pip install accelerate`)z/No GPU found. A GPU is needed for quantization.Ú
device_mapzŽYou have loaded an EETQ model on CPU and have a CUDA device available, make sure to set your model on a GPU device in order to run your model.r   ÚcpuÚdiskz¯You are attempting to load an EETQ model with a device_map that contains a CPU or disk device. This is not supported. Please remove the CPU or disk device from the device_map.)r
   ÚImportErrorr	   ÚtorchÚcudaÚis_availableÚRuntimeErrorÚgetÚloggerÚwarning_onceÚ
isinstanceÚdictÚlenÚvaluesÚ
ValueError)r   Úargsr   r   s       r   Úvalidate_environmentÚ$EetqHfQuantizer.validate_environment-   sÎ   € Ü#×%Ñ%ÜÐhÓiÐiä&×(Ñ(ÜÐnÓoÐoä�z‰z×&Ñ&×(Ñ(ÜÐPÓQÐQà—Z‘Z Ó-ˆ
ØÑÜ×ÑðIõô ˜
¤D×)Ñ)Ü�:‹ Ó" u°
×0AÑ0AÓ0CÓ'CÀvÐQ[×QbÑQbÓQdÓGdÜ ðhóð ð Heð *r   Úreturnc                 óX   • U[         R                  :w  a  [        R                  S5        U$ )NzLWe suggest you to set `dtype=torch.float16` for better efficiency with EETQ.)r    Úfloat16r%   Úinfo)r   Údtypes     r   Úupdate_dtypeÚEetqHfQuantizer.update_dtypeD   s    € Ø”E—M‘MÓ!Ü�K‰KÐfÔgØˆr   Úmodelr   Ú
param_namec                 ó|   • SSK Jn  [        X5      u  pV[        XT5      (       a  U R                  (       d  US:X  a  ggg)Nr   )Ú
EetqLinearÚbiasFT)Úintegrations.eetqr9   r   r'   Úpre_quantized)r   r6   r7   r   r9   ÚmoduleÚtensor_names          r   Úparam_needs_quantizationÚ(EetqHfQuantizer.param_needs_quantizationI   s6   € Ý2ä2°5ÓEÑˆä�f×)Ñ)Ø×!×! [°FÓ%:ØàØr   c                 ó°   • SSK Jn  U R                  XR                  R                  UR
                  5      U l        U" XR                  U R                  S9ng )Nr   )Úreplace_with_eetq_linear)Úmodules_to_not_convertr<   )ÚintegrationsrB   Úget_modules_to_not_convertr   rC   Ú_keep_in_fp32_modulesr<   )r   r6   r   rB   s       r   Ú$_process_model_before_weight_loadingÚ4EetqHfQuantizer._process_model_before_weight_loadingU   sO   € õ
 	<à&*×&EÑ&EØ×+Ñ+×BÑBÀE×D_ÑD_ó'
ˆÔ#ñ )Ø×*EÑ*EÐUY×UgÑUgñ
‰r   c                 ó   • g©NT© ©r   s    r   Úis_serializableÚEetqHfQuantizer.is_serializabled   s   € Ør   c                 ó   • grJ   rK   rL   s    r   Úis_trainableÚEetqHfQuantizer.is_trainableg   s   € àr   c                 ó   • SSK Jn  U" U 5      $ )Nr   )ÚEetqQuantize)r;   rS   )r   rS   s     r   Úget_quantize_opsÚ EetqHfQuantizer.get_quantize_opsk   s   € Ý4á˜DÓ!Ð!r   )rC   )r3   útorch.dtyper/   rV   )r6   r   )Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__Úrequires_calibrationÚ__annotations__r   r-   r4   ÚstrÚboolr?   rG   rM   ÚpropertyrP   rT   Ú__static_attributes__Ú__classcell__)r   s   @r   r   r   "   sx   ø‡ ñð !ÐØ%Ó%õ8òô.ð

Ð.?ð 
ÈSð 
Ð_cô 
ð
à ô
òð ð˜dó ó ð÷"ð "r   r   )Útypingr   Úbaser   Úmodeling_utilsr   Úutils.quantization_configr   Úutilsr	   r
   r   r   Úquantizers_utilsr   r    Ú
get_loggerrW   r%   r   rK   r   r   Ú<module>rj      sO   ðõ !å ö Ý0Ý6ç ^Ó ^Ý 2ñ ×ÑÛð 
×	Ò	˜HÓ	%€ôL"�kõ L"r   