ó
    qyüiµ  ã                   ó°   • S SK Jr  SSKJr  \(       a  SSKJr  SSKJr  SSKJ	r	J
r
Jr  \
" 5       (       a  S SKr\R                  " \5      r " S	 S
\5      rg)é    )ÚTYPE_CHECKINGé   )ÚHfQuantizeré   )ÚPreTrainedModel)ÚBitNetQuantConfig)Úis_accelerate_availableÚis_torch_availableÚloggingNc                   ó¾   ^ • \ rS rSr% SrSrS\S'   U 4S jrS r  SS jr	S	\
\\\-  4   S
\
\\\-  4   4S jrS r\S
\4S j5       r\S
\4S j5       rS rSrU =r$ )ÚBitNetHfQuantizeré!   zã
1.58-bit quantization from BitNet quantization method:
Before loading: it converts the linear layers into BitLinear layers during loading.

Check out the paper introducing this method: https://huggingface.co/papers/2402.17764
Tr   Úquantization_configc                 ó(   >• [         TU ]  " U40 UD6  g )N)ÚsuperÚ__init__)Úselfr   ÚkwargsÚ	__class__s      €Úe/home/mande/repo/quber/.venv/lib/python3.13/site-packages/transformers/quantizers/quantizer_bitnet.pyr   ÚBitNetHfQuantizer.__init__,   s   ø€ Ü‰ÒÐ,Ñ7°Ó7ó    c                 ó®  • [        5       (       d  [        S5      e[        R                  R	                  5       (       d  [
        R                  S5        g UR                  S5      nUc  [
        R                  S5        g [        U[        5      (       aC  [        U5      S:”  a  SUR                  5       ;   d  SUR                  5       ;   a  [        S5      eg g )	NzOLoading a BitNet quantized model requires accelerate (`pip install accelerate`)zhYou don't have a GPU available to load the model, the inference will be slow because of weight unpackingÚ
device_mapz�You have loaded a BitNet model on CPU and have a CUDA device available, make sure to set your model on a GPU device in order to run your model.r   ÚcpuÚdiskz¯You are attempting to load a BitNet model with a device_map that contains a CPU or disk device.This is not supported. Please remove the CPU or disk device from the device_map.)r	   ÚImportErrorÚtorchÚcudaÚis_availableÚloggerÚwarning_onceÚgetÚ
isinstanceÚdictÚlenÚvaluesÚ
ValueError)r   Úargsr   r   s       r   Úvalidate_environmentÚ&BitNetHfQuantizer.validate_environment/   sÃ   € Ü&×(Ñ(ÜÐoÓpÐpä�z‰z×&Ñ&×(Ñ(Ü×ÑØzôð à—Z‘Z Ó-ˆ
ØÑÜ×ÑðIõô ˜
¤D×)Ñ)Ü�:‹ Ó" u°
×0AÑ0AÓ0CÓ'CÀvÐQ[×QbÑQbÓQdÓGdÜ ðgóð ð Heð *r   c                 ó²   • SSK Jn  U R                  XR                  R                  UR
                  5      U l        U" UU R                  U R                  S9ng )Nr   )Úreplace_with_bitnet_linear)Úmodules_to_not_convertr   )Úintegrationsr-   Úget_modules_to_not_convertr   r.   Ú_keep_in_fp32_modules)r   Úmodelr   r-   s       r   Ú$_process_model_before_weight_loadingÚ6BitNetHfQuantizer._process_model_before_weight_loadingF   sR   € õ
 	>à&*×&EÑ&EØ×+Ñ+×BÑBÀE×D_ÑD_ó'
ˆÔ#ñ +ØØ#'×#>Ñ#>Ø $× 8Ñ 8ñ
‰r   Ú
max_memoryÚreturnc                 ó`   • UR                  5        VVs0 s H
  u  p#X#S-  _M     nnnU$ s  snnf )NgÍÌÌÌÌÌì?)Úitems)r   r5   ÚkeyÚvals       r   Úadjust_max_memoryÚ#BitNetHfQuantizer.adjust_max_memoryW   s5   € Ø6@×6FÑ6FÔ6HÔIÒ6H©(¨#�c ™:’oÑ6Hˆ
ÑIØÐùó Js   ”*c                 ó   • g)NT© ©r   s    r   Úis_serializableÚ!BitNetHfQuantizer.is_serializable[   s   € Ør   c                 ót   • U R                   R                  S:H  =(       a    U R                   R                  S:H  $ )NÚautobitlinearÚonline©r   Úlinear_classÚquantization_moder?   s    r   Úis_trainableÚBitNetHfQuantizer.is_trainable^   s7   € ð ×$Ñ$×1Ñ1°_ÑD÷ GØ×(Ñ(×:Ñ:¸hÑFð	
r   c                 ót   • U R                   R                  S:H  =(       a    U R                   R                  S:H  $ )zUFlag indicating whether the quantized model can carry out quantization aware trainingrC   rD   rE   r?   s    r   Úis_qat_trainableÚ"BitNetHfQuantizer.is_qat_trainablee   s7   € ð ×$Ñ$×1Ñ1°_ÑD÷ GØ×(Ñ(×:Ñ:¸hÑFð	
r   c                 óª   • SSK Jn  SSKJn  U R                  R
                  S:X  a,  U R                  R                  S:X  a  U" S/S/U" U 5      /S9/$ / $ )Nr   )ÚWeightConverter)ÚBitNetDeserializerC   ÚofflineÚweight)Úsource_patternsÚtarget_patternsÚ
operations)Úcore_model_loadingrN   Úintegrations.bitnetrO   r   rF   rG   )r   rN   rO   s      r   Úget_weight_conversionsÚ(BitNetHfQuantizer.get_weight_conversionsm   sb   € Ý8Ý;ð ×$Ñ$×1Ñ1°_ÓDØ×(Ñ(×:Ñ:¸iÓGñ  Ø%- JØ%- JÙ 1°$Ó 7Ð8ñðð ð ˆ	r   )r.   )r2   r   )Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__Úrequires_calibrationÚ__annotations__r   r*   r3   r%   ÚstrÚintr;   r@   ÚpropertyÚboolrH   rK   rW   Ú__static_attributes__Ú__classcell__)r   s   @r   r   r   !   s    ø‡ ñð  ÐØ,Ó,õ8òð.
à ô
ð"¨D°°c¸C±i°Ñ,@ð ÀTÈ#ÈsÐUXÉyÈ.ÑEYô òð ð
˜dó 
ó ð
ð ð
 $ó 
ó ð
÷ð r   r   )Útypingr   Úbaser   Úmodeling_utilsr   Úutils.quantization_configr   Úutilsr	   r
   r   r   Ú
get_loggerrY   r!   r   r>   r   r   Ú<module>rl      sL   ðõ !å ö Ý0Ý=ç HÑ Hñ ×ÑÛð 
×	Ò	˜HÓ	%€ô[˜õ [r   