ó
    qyüiØ  ã                   óÔ   • S SK JrJr  SSKJr  SSKJr  \(       a  SSKJr  SSK	J
r
  SSKJrJrJrJrJr  SS	K	Jr  \" 5       (       a  S S
Kr\R&                  " \5      r " S S\5      rg
)é    )ÚTYPE_CHECKINGÚOptionalé   )ÚHfQuantizer)Úget_module_from_nameé   )ÚPreTrainedModel)ÚFPQuantConfig)Úis_fp_quant_availableÚis_qutlass_availableÚis_torch_availableÚis_torch_xpu_availableÚlogging)ÚQuantizationConfigMixinNc                   ó°   ^ • \ rS rSr% SrSrSrS\S'   S\4U 4S jjr	S r
SS
 jrSSS\S	\4S jr  SS jr\SS\S   4S jj5       rS rS rS rSrU =r$ )ÚFPQuantHfQuantizeré"   z„
Quantizer for the FP-Quant method. Enables the loading of prequantized models and in-flight quantization of full-precision models.
FTr
   Úquantization_configc                 ó(   >• [         TU ]  " U40 UD6  g ©N)ÚsuperÚ__init__)Úselfr   ÚkwargsÚ	__class__s      €Úg/home/mande/repo/quber/.venv/lib/python3.13/site-packages/transformers/quantizers/quantizer_fp_quant.pyr   ÚFPQuantHfQuantizer.__init__+   s   ø€ Ü‰ÒÐ,Ñ7°Ó7ó    c                 óÄ  • [         R                  R                  5       (       d  [        5       (       d  [	        S5      e[        5       (       d&  U R                  R                  (       d  [        S5      eU R                  R                  (       am  U R                  R                  S:X  aS  [         R                  R                  5       (       a0  [         R                  R                  5       S   S:  a  [        S5      eU R                  R                  (       a  [        R                  S5        [        5       (       d  [        S5      eUc&  U R                  R                  (       d  [        S	5      e[        U[         5      (       a^  U R                  R                  (       d#  [#        U5      S
:”  a  SUR%                  5       ;   d  SUR%                  5       ;   a  [        S5      eg g )Nz]FPQuant quantization is only supported on GPU or Intel XPU. Please use a different quantizer.a€  Using `fp_quant` with real quantization requires a **Blackwell GPU** and qutlass: `git clone https://github.com/IST-DASLab/qutlass.git && cd qutlass && pip install --no-build-isolation .`. You can use `FPQuantConfig(pseudoquantization=True, ...)` to use Triton-based pseudo-quantization. It doesn't provide any speedups but emulates the quantization behavior of the real quantization.Únvfp4r   é	   zäNVFP4 pseudoquantization requires a GPU with compute capability >= 9.0 (Hopper or newer) because the Triton kernel uses the `fp8e4nv` type. Please use `forward_dtype='mxfp4'` instead, or use a GPU with compute capability >= 9.0.zŠUsing pseudo-quantization for FP-Quant. This doesn't provide any speedups but emulates the quantization behavior of the real quantization.zGUsing `fp_quant` quantization requires fp_quant: `pip install fp_quant`zyYou are attempting to load a FPQuant model without setting device_map. Please set device_map comprised of 'cuda' devices.r   ÚcpuÚdiskz±You are attempting to load a FPQuant model with a device_map that contains a CPU or disk device. This is not supported. Please remove the CPU or disk device from the device_map.)ÚtorchÚcudaÚis_availabler   ÚNotImplementedErrorr   r   ÚpseudoquantizationÚImportErrorÚforward_dtypeÚget_device_capabilityÚ
ValueErrorÚloggerÚwarningr   Ú
isinstanceÚdictÚlenÚvalues)r   Ú
device_mapr   s      r   Úvalidate_environmentÚ'FPQuantHfQuantizer.validate_environment.   sˆ  € Ü�z‰z×&Ñ&×(Ñ(Ô1G×1IÑ1IÜ%Øoóð ô $×%Ñ%¨d×.FÑ.F×.Y×.YÜð Sóð ð
 ×$Ñ$×7×7Ø×(Ñ(×6Ñ6¸'ÓAÜ—
‘
×'Ñ'×)Ñ)Ü—
‘
×0Ñ0Ó2°1Ñ5¸Ó9äð?óð ð ×#Ñ#×6×6Ü�N‰Nð ]ôô %×&Ñ&ÜÐgÓhÐhàÑ d×&>Ñ&>×&Q×&QÜðFóð ô ˜
¤D×)Ñ)à×,Ñ,×?×?Ü˜
“O aÓ'Ø˜Z×.Ñ.Ó0Ó0Ø˜Z×.Ñ.Ó0Ó0ä ðhóð ð 1ð *r   Úreturnc                 ó€   • U[         R                  :w  a)  [        R                  SU S35        [         R                  nU$ )NzSetting dtype to zP, but only bfloat16 is supported right now. Overwriting torch_dtype to bfloat16.)r$   Úbfloat16r-   Úwarning_once)r   Údtypes     r   Úupdate_dtypeÚFPQuantHfQuantizer.update_dtype^   s9   € Ø”E—N‘NÓ"Ü×ÑØ# E 7Ð*zÐ{ôô —N‘NˆEØˆr   Úmodelr	   Ú
param_namec                 óX   • SSK Jn  [        X5      u  pV[        XT5      (       a  US;   a  gg)Nr   )ÚFPQuantLinear)ÚweightÚqweightÚdqweightTF)Úfp_quantr@   r   r/   )r   r=   r>   r   r@   ÚmoduleÚtensor_names          r   Úparam_needs_quantizationÚ+FPQuantHfQuantizer.param_needs_quantizationf   s+   € Ý*ä2°5ÓEÑˆÜ�f×,Ñ,°Ð@aÓ1aààr   c                 óJ   • SSK Jn  SSKJn  U" UU" U R                  5      S9  g )Nr   )Úreplace_with_fp_quant_linearr   )Úadapt_fp_quant_config)Úfp_quant_linear_config)rD   rJ   Úintegrations.fp_quantrK   r   )r   r=   r   rJ   rK   s        r   Ú$_process_model_before_weight_loadingÚ7FPQuantHfQuantizer._process_model_before_weight_loadingp   s#   € õ
 	:åAá$ØÙ#8¸×9QÑ9QÓ#Ró	
r   c                 ój   • U R                   R                  nU(       d  [        R                  S5        U$ )Nz²You are attempting to train a model with FPQuant quantization. This is only supported when `store_master_weights=True`. Please set `store_master_weights=True` to train the model.)r   Ústore_master_weightsr-   r.   )r   r=   Ú	trainables      r   Úis_trainableÚFPQuantHfQuantizer.is_trainable~   s0   € à×,Ñ,×AÑAˆ	ÞÜ�N‰Nð Eôð Ðr   c                 ó   • g)NT© )r   s    r   Úis_serializableÚ"FPQuantHfQuantizer.is_serializable‡   s   € Ør   c                 ó   • SSK Jn  U" U 5      $ )Nr   )ÚFpQuantQuantize)rM   rZ   )r   rZ   s     r   Úget_quantize_opsÚ#FPQuantHfQuantizer.get_quantize_opsŠ   s   € Ý;á˜tÓ$Ð$r   c                 óº   • SSK Jn  SSKJn  U R                  (       a=  U R
                  R                  (       a  U" S/SU" U 5      /S9/$ U" S/SU" U 5      /S9/$ / $ )Nr   )ÚWeightConverter)ÚFpQuantDeserializez	.dqweight)Úsource_patternsÚtarget_patternsÚ
operationsz.qweight)Úcore_model_loadingr^   rM   r_   Úpre_quantizedr   r(   )r   r^   r_   s      r   Úget_weight_conversionsÚ)FPQuantHfQuantizer.get_weight_conversions�   ss   € Ý8Ý>à××Ø×'Ñ'×:×:á#Ø)4¨Ø(3Ù$6°tÓ$<Ð#=ñðð ñ $Ø)3¨Ø(2Ù$6°tÓ$<Ð#=ñðð ð ˆ	r   rV   )r:   útorch.dtyper6   rg   )r=   r	   r   )Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__Úrequires_calibrationÚis_qat_trainableÚ__annotations__r   r   r4   r;   ÚstrÚboolrG   rN   Úpropertyr   rS   rW   r[   re   Ú__static_attributes__Ú__classcell__)r   s   @r   r   r   "   s’   ø‡ ñð !ÐØÐØ(Ó(ð8Ð,C÷ 8ò.ô`ðÐ.?ð ÈSð Ð_cô ð
à ô
ð ñ (Ð+<Ñ"=ô ó ðòò%÷
ð r   r   )Útypingr   r   Úbaser   Úquantizers_utilsr   Úmodeling_utilsr	   Úutils.quantization_configr
   Úutilsr   r   r   r   r   r   r$   Ú
get_loggerrh   r-   r   rV   r   r   Ú<module>r|      sP   ð÷ +å Ý 2ö Ý0Ý9ç tÕ tÝ ?ñ ×ÑÛà	×	Ò	˜HÓ	%€ôB˜õ Br   