ó
    qyüià  ã                   óŠ   • S SK JrJrJr  S SKJr  SSKJr  \" 5       (       a  SSKr\R                  " \
5      r " S S\5      rg)	é   )Úis_compressed_tensors_availableÚis_torch_availableÚlogging)ÚCompressedTensorsConfigé   )ÚHfQuantizeré    Nc                   ó’   ^ • \ rS rSr% SrSr\\S'   S\4U 4S jjrS r	SS jr
S	 rS
 rS r\S 5       rS\4S jrS\4S jrSrU =r$ )ÚCompressedTensorsHfQuantizeré   zu
Quantizer for the compressed_tensors package.  Loads and restores models to
quantized state with compressed_tensors
TÚquantization_configc                 óâ   >• [         TU ]  " U40 UD6  [        5       (       d  [        S5      eUR	                  5         SSKJn  UR                  U5      U l        UR                  U l	        Xl
        g )NúuUsing `compressed_tensors` quantized models requires the compressed-tensors library: `pip install compressed-tensors`r	   )ÚModelCompressor)ÚsuperÚ__init__r   ÚImportErrorÚ	post_initÚcompressed_tensors.compressorsr   Úfrom_compression_configÚ
compressorÚrun_compressedr   )Úselfr   Úkwargsr   Ú	__class__s       €Úq/home/mande/repo/quber/.venv/lib/python3.13/site-packages/transformers/quantizers/quantizer_compressed_tensors.pyr   Ú%CompressedTensorsHfQuantizer.__init__$   si   ø€ Ü‰ÒÐ,Ñ7°Ò7ä.×0Ñ0Üð3óð ð 	×%Ñ%Ô'ÝBà)×AÑAÐBUÓVˆŒØ1×@Ñ@ˆÔØ#6Õ ó    c                 ó8   • [        5       (       d  [        S5      eg )Nr   )r   r   )r   Úargsr   s      r   Úvalidate_environmentÚ1CompressedTensorsHfQuantizer.validate_environment7   s"   € Ü.×0Ñ0Üð3óð ð 1r   Úreturnc                 óX   • U[         R                  :w  a  [        R                  S5        U$ )NzZWe suggest you to set `dtype=torch.float16` for better efficiency with compressed_tensors.)ÚtorchÚfloat16ÚloggerÚinfo)r   Údtypes     r   Úupdate_dtypeÚ)CompressedTensorsHfQuantizer.update_dtype>   s    € Ø”E—M‘MÓ!Ü�K‰KÐtÔuØˆr   c                 ó  • SSK Jn  U R                  R                  nU" XU R                  5        U R                  R
                  (       d  U R                  R                  (       a  U R                  R                  US9  g g )Nr	   )Úapply_quantization_config©Úmodel)Úcompressed_tensors.quantizationr-   r   r   r   Úis_quantization_compressedÚis_sparsification_compressedÚcompress_model)r   r/   r   r-   Úct_quantization_configs        r   Ú$_process_model_before_weight_loadingÚACompressedTensorsHfQuantizer._process_model_before_weight_loadingC   s`   € ÝMà!%§¡×!DÑ!DÐñ 	" %À×ATÑATÔUà×$Ñ$×?×?Ø×'Ñ'×D×Dà�O‰O×*Ñ*°Ð*Ò7ð Er   c                 óÆ   • U R                   R                  (       a  U R                  (       a  U R                   R                  (       a  U R                  R                  US9  gg)z3Decompress loaded model if necessary - need for qatr.   N)r   r1   r   r2   r   Údecompress_model)r   r/   r   s      r   Ú#_process_model_after_weight_loadingÚ@CompressedTensorsHfQuantizer._process_model_after_weight_loadingP   sE   € ð ×$Ñ$×?×?È×H[×H[Ø×$Ñ$×A×Aà�O‰O×,Ñ,°5Ð,Ò9ð Br   c                 óÀ   • SSSSSS.nUR                  5       bD  UR                  5       R                  b)  UR                  5       R                  R                  U5        U$ )NÚcolwiseÚrowwise)z0layers.*.feed_forward.experts.*.gate_proj.weightz6layers.*.feed_forward.experts.*.gate_proj.weight_scalez.layers.*.feed_forward.experts.*.up_proj.weightz4layers.*.feed_forward.experts.*.up_proj.weight_scalez0layers.*.feed_forward.experts.*.down_proj.weight)Úget_text_configÚbase_model_tp_planÚupdate)r   ÚconfigÚadditional_plans      r   Úupdate_tp_planÚ+CompressedTensorsHfQuantizer.update_tp_planZ   s_   € à@IØFOØ>GØDMØ@Iñ
ˆð ×!Ñ!Ó#Ñ/°F×4JÑ4JÓ4L×4_Ñ4_Ñ4kØ×"Ñ"Ó$×7Ñ7×>Ñ>¸ÔOàˆr   c                 ó   • g)NT© ©r   s    r   Úis_trainableÚ)CompressedTensorsHfQuantizer.is_trainableg   ó   € àr   c                 óh   • U R                   (       + =(       d    U R                  R                  (       + $ )z7Loaded Models can carry out quantization aware training)r   r   r1   rG   s    r   Úis_qat_trainableÚ-CompressedTensorsHfQuantizer.is_qat_trainablek   s'   € ð ×&Ñ&Ô&×a¨d×.FÑ.F×.aÑ.aÔ*aÐar   c                 ó   • g)z>Models quantized using compressed tensors can be saved to diskTrF   rG   s    r   Úis_serializableÚ,CompressedTensorsHfQuantizer.is_serializablep   rJ   r   )r   r   r   )r)   útorch.dtyper#   rQ   )Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__Úrequires_calibrationr   Ú__annotations__r   r!   r*   r5   r9   rC   ÚpropertyrH   ÚboolrL   rO   Ú__static_attributes__Ú__classcell__)r   s   @r   r   r      so   ø‡ ñð
  ÐØ0Ó0ð7Ð,C÷ 7ò&ôò
8ò:òð ñó ððb $ô bð
 ÷ ò r   r   )Úutilsr   r   r   Úutils.quantization_configr   Úbaser   r%   Ú
get_loggerrR   r'   r   rF   r   r   Ú<module>ra      s@   ð÷  QÑ PÝ ?Ý ñ ×ÑÛà	×	Ò	˜HÓ	%€ôW ;õ Wr   