ó
    pyüií‚  ã                   ó|  • S r SSKrSSKrSSKrSSKrSSKJr  SSKJ	r	  SSK
JrJrJr  \" 5       (       a
  SSKrSSKJr  \R                   " \5      rS r\" 5       (       a  \" 5       (       a  SS	KJr  OSS
KJr   " S S\5      r " S S\5      rSqS rS rS rS rS rS r SS jr!S r"SS jr#SS jr$SS jr%S r&g)z
Integration with Deepspeed
é    N)Úpartialmethodé   )Údep_version_check)Úis_accelerate_availableÚis_torch_availableÚlogging)Únnc                  óÞ   • [         R                  R                  S5      S Ln U (       a!   [         R                  R                  S5      ngg ! [         R                  R                   a     gf = f)NÚ	deepspeedTF)Ú	importlibÚutilÚ	find_specÚmetadataÚPackageNotFoundError)Úpackage_existsÚ_s     Ú`/home/mande/repo/quber/.venv/lib/python3.13/site-packages/transformers/integrations/deepspeed.pyÚis_deepspeed_availabler   $   sc   € Ü—^‘^×-Ñ-¨kÓ:À$ÐF€Nö ð	Ü×"Ñ"×+Ñ+¨KÓ8ˆAØð øô ×!Ñ!×6Ñ6ó 	Ùð	ús   ªA ÁA,Á+A,)ÚHfDeepSpeedConfig)Úobjectc                   ó,   ^ • \ rS rSrSrU 4S jrSrU =r$ )r   é9   a"  
This object contains a DeepSpeed configuration dictionary and can be quickly queried for things like zero stage.

A `weakref` of this object is stored in the module's globals to be able to access the config from areas where
things like the Trainer object is not available (e.g. `from_pretrained` and `_get_resized_embeddings`). Therefore
it's important that this object remains alive while the program is still running.

[`Trainer`] uses the `HfTrainerDeepSpeedConfig` subclass instead. That subclass has logic to sync the configuration
with values of [`TrainingArguments`] by replacing special placeholder values: `"auto"`. Without this special logic
the DeepSpeed configuration is not modified in any way.

Args:
    config_file_or_dict (`Union[str, Dict]`): path to DeepSpeed config file or dict.

c                 óf   >• [        U 5        [        S5        [        S5        [        TU ]  U5        g )NÚ
accelerater   )Úset_hf_deepspeed_configr   ÚsuperÚ__init__©ÚselfÚconfig_file_or_dictÚ	__class__s     €r   r   ÚHfDeepSpeedConfig.__init__J   s)   ø€ ä Ô%Ü˜,Ô'Ü˜+Ô&Ü‰ÑÐ,Õ-ó    © )Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__r   Ú__static_attributes__Ú__classcell__©r!   s   @r   r   r   9   s   ø† ñ÷ .ó .r#   r   c                   ó`   ^ • \ rS rSrSrU 4S jrS rS rSS jr\	" \SS9r
SS	 jrS
 rSrU =r$ )ÚHfTrainerDeepSpeedConfigéR   z’
The `HfTrainerDeepSpeedConfig` object is meant to be created during `TrainingArguments` object creation and has the
same lifespan as the latter.
c                 ó@   >• [         TU ]  U5        S U l        / U l        g ©N)r   r   Ú_dtypeÚ
mismatchesr   s     €r   r   Ú!HfTrainerDeepSpeedConfig.__init__X   s   ø€ Ü‰ÑÐ,Ô-ØˆŒØˆ�r#   c                 óJ   • U R                   c  [        S5      eU R                   $ )Nz8trainer_config_process() wasn't called yet to tell dtype)r2   Ú
ValueError)r   s    r   ÚdtypeÚHfTrainerDeepSpeedConfig.dtype]   s"   € Ø�;‰;ÑÜÐWÓXÐXØ�{‰{Ðr#   c                 ó6   • U R                  U5      nUc  gUS:H  $ )NFÚauto)Ú	get_value)r   Úds_key_longÚvals      r   Úis_autoÚ HfTrainerDeepSpeedConfig.is_autob   s"   € Ø�n‰n˜[Ó)ˆØ‰;Øà˜&‘=Ð r#   c           
      óú   • U R                  U5      u  pVUc  gUR                  U5      S:X  a  X%U'   gU(       d  gUR                  U5      nUb.  Xr:w  a(  U R                  R                  SU SU SU SU 35        ggg)a†  
A utility method that massages the config file and can optionally verify that the values match.

1. Replace "auto" values with `TrainingArguments` value.

2. If it wasn't "auto" and `must_match` is true, then check that DS config matches Trainer
config values and if mismatched add the entry to `self.mismatched` - will assert during
`trainer_config_finalize` for one or more mismatches.

Nr:   z- ds Ú=z vs hf )Úfind_config_nodeÚgetr3   Úappend)r   r<   Úhf_valÚhf_keyÚ
must_matchÚconfigÚds_keyÚds_vals           r   Ú
fill_matchÚ#HfTrainerDeepSpeedConfig.fill_matchi   s�   € ð ×.Ñ.¨{Ó;‰ˆØ‰>Øà�:‰:�fÓ Ó'Ø#�6‰NØæØà—‘˜FÓ#ˆØÑ &Ó"2Ø�O‰O×"Ñ" U¨;¨-°q¸¸ÀÈÀxÈqÐQWÐPXÐ#YÕZð #3Ðr#   F)rG   c                 óà  • UR                   UR                  -  UR                  -  nU R                  SUR                  SU(       + 5        U R                  SUR                  S5        U R                  SUSU(       + 5        U R                  SUR                  S5        U R                  SUR
                  S	5        U R                  S
UR                  UR                  /S5        U R                  SUR                  S5        U R                  SUR                  S5        U R                  SS5        U R                  SUR
                  S	5        UR                  (       aE  U R                  R                  S0 5      U R                  S'   UR                  U R                  S   S'   U R                  SUR                  =(       d    UR                  S5        U R                  SUR                   =(       d    UR"                  S5        U R%                  S5      (       a  [&        R(                  U l        gU R%                  S5      (       a  [&        R,                  U l        g[&        R.                  U l        g)zr
Adjust the config with `TrainingArguments` values. This stage is run during `TrainingArguments` object
creation.
Útrain_micro_batch_size_per_gpuÚper_device_train_batch_sizeÚgradient_accumulation_stepsÚtrain_batch_sizeztrain_batch_size (calculated)Úgradient_clippingÚmax_grad_normzoptimizer.params.lrÚlearning_ratezoptimizer.params.betaszadam_beta1+adam_beta2zoptimizer.params.epsÚadam_epsilonzoptimizer.params.weight_decayÚweight_decayzscheduler.params.warmup_min_lrr   zscheduler.params.warmup_max_lrÚ
checkpointÚuse_node_local_storagezfp16.enabledzfp16|fp16_full_evalzbf16.enabledzbf16|bf16_full_evalN)Ú
world_sizerO   rP   rK   rS   rT   Ú
adam_beta1Ú
adam_beta2rU   rV   Ú	fill_onlyÚsave_on_each_noderH   rC   Úfp16Úfp16_full_evalÚbf16Úbf16_full_evalÚis_trueÚtorchÚbfloat16r2   Úfloat16Úfloat32)r   ÚargsÚauto_find_batch_sizerQ   s       r   Útrainer_config_processÚ/HfTrainerDeepSpeedConfig.trainer_config_process…   sí  € ð  Ÿ?™?¨T×-MÑ-MÑMÐPT×PpÑPpÑpÐØ�‰Ø,Ø×,Ñ,Ø)Ø$Ô$ô		
ð 	�‰Ø)Ø×,Ñ,Ø)ô	
ð
 	�‰ØØØ+Ø$Ô$ô		
ð 	�‰Ð+¨T×-?Ñ-?ÀÔQà�‰Ð-¨t×/AÑ/AÀ?ÔSØ�‰Ø$Ø�_‰_˜dŸo™oÐ.Ø#ô	
ð
 	�‰Ð.°×0AÑ0AÀ>ÔRØ�‰Ð7¸×9JÑ9JÈNÔ[à�‰Ð7¸Ô;Ø�‰Ð8¸$×:LÑ:LÈoÔ^ð ×!×!à(,¯©¯©¸ÀbÓ(IˆD�K‰K˜Ñ%ØBF×BXÑBXˆD�K‰K˜Ñ%Ð&>Ñ?ð 	�‰˜¨¯©×)I°d×6IÑ6IÐLaÔbØ�‰˜¨¯©×)I°d×6IÑ6IÐLaÔbð �<‰<˜×'Ñ'ÜŸ.™.ˆD�KØ�\‰\˜.×)Ñ)ÜŸ-™-ˆD�KäŸ-™-ˆD�Kr#   c                 óò  • / SQnU Vs/ s H  oPR                  U5      (       d  M  UPM     nn[        U5      S:”  Ga½  Sn[        US5      (       Ga8  [        UR                  S5      (       a  UR                  R                  nGO[        UR                  S5      (       a   [        UR                  R                  5      nOÊ[        UR                  S5      (       aF  [        UR                  R                  S5      (       a!  UR                  R                  R                  nOi[        UR                  S5      (       aN  [        UR                  R                  S5      (       a)  [        UR                  R                  R                  5      nUc  [        SU S	35      eU R                  S
Xw-  5        U R                  5       (       a6  U R                  S[        SU-  U-  5      5        U R                  SSU-  5        U R                  SUS5        U R                  SUR                  U5      S5        [        U R                  5      S:”  a*  SR                  U R                  5      n[        SU S35      egs  snf )zx
This stage is run after we have the model and know num_training_steps.

Now we can complete the configuration process.
)ú$zero_optimization.reduce_bucket_sizeú-zero_optimization.stage3_prefetch_bucket_sizeú4zero_optimization.stage3_param_persistence_thresholdr   NrH   Úhidden_sizeÚhidden_sizesÚtext_configz½The model's config file has neither `hidden_size` nor `hidden_sizes` entry, therefore it's not possible to automatically fill out the following `auto` entries in the DeepSpeed config file: zb. You can fix that by replacing `auto` values for these keys with an integer value of your choice.rl   rm   gÍÌÌÌÌÌì?rn   é
   z scheduler.params.total_num_stepsznum_training_steps (calculated)z!scheduler.params.warmup_num_stepsÚwarmup_stepsÚ
z]Please correct the following DeepSpeed config values that mismatch TrainingArguments values:
zF
The easiest method is to set these DeepSpeed config values to 'auto'.)r>   ÚlenÚhasattrrH   ro   Úmaxrp   rq   r6   r\   Úis_zero3ÚintrK   Úget_warmup_stepsr3   Újoin)	r   rg   ÚmodelÚnum_training_stepsÚhidden_size_based_keysÚxÚhidden_size_auto_keysro   r3   s	            r   Útrainer_config_finalizeÚ0HfTrainerDeepSpeedConfig.trainer_config_finalize¿   s   € ò"
Ðñ
 -CÓ VÒ,B qÇlÁlÐSTÇo§Ñ,BÐÐ VäÐ$Ó%¨Ô)ØˆKÜ�u˜h×'Ò'Ü˜5Ÿ<™<¨×7Ñ7Ø"'§,¡,×":Ñ":’KÜ˜UŸ\™\¨>×:Ñ:ä"% e§l¡l×&?Ñ&?Ó"@‘KÜ˜UŸ\™\¨=×9Ñ9¼gÀeÇlÁl×F^ÑF^Ð`m×>nÑ>nØ"'§,¡,×":Ñ":×"FÑ"F‘KÜ˜UŸ\™\¨=×9Ñ9¼gÀeÇlÁl×F^ÑF^Ð`n×>oÑ>oä"% e§l¡l×&>Ñ&>×&KÑ&KÓ"L�KàÑ"Ü ð5à5JÐ4Kð LYðYóð ð �N‰NÐAÀ;ÑC\Ô]Ø�}‰}�‰à—‘ØCÜ˜˜kÑ)¨KÑ7Ó8ôð —‘ØJØ˜Ñ$ôð 	�‰Ø.ØØ-ô	
ð
 	�‰Ø/Ø×!Ñ!Ð"4Ó5Øô	
ô ˆt�‰Ó !Ó#ØŸ™ 4§?¡?Ó3ˆJÜðØ'˜LÐ(oðqóð ð $ùòa !Ws
   ‰I4¦I4)r2   r3   )NT©F)r%   r&   r'   r(   r)   r   r7   r>   rK   r   r\   ri   r�   r*   r+   r,   s   @r   r.   r.   R   s=   ø† ñõ
ò
ò
!ô[ñ4 ˜j°UÑ;€Iô8(÷tCð Cr#   r.   c                 ó0   • [         R                  " U 5      qg r1   )ÚweakrefÚrefÚ_hf_deepspeed_config_weak_ref)Úhf_deepspeed_config_objs    r   r   r   	  s   € ô
 %,§K¢KÐ0GÓ$HÑ!r#   c                  ó   • S q g r1   )r‡   r$   r#   r   Úunset_hf_deepspeed_configrŠ     s
   € ð %)Ñ!r#   c                  óX   • [         b#  [        5       b  [        5       R                  5       $ g)NF)r‡   rx   r$   r#   r   Úis_deepspeed_zero3_enabledrŒ     s&   € Ü$Ñ0Ô5RÓ5TÑ5`Ü,Ó.×7Ñ7Ó9Ð9àr#   c                  óP   • [         b  [        5       b  [        5       R                  $ g r1   )r‡   rH   r$   r#   r   Údeepspeed_configrŽ     s#   € Ü$Ñ0Ô5RÓ5TÑ5`Ü,Ó.×5Ñ5Ð5àr#   c                 ó"  ^^^^• SSK mSSKnSSKJn  SSKJm  U R                  5       mUUUU4S jmUR                  " 5          U" 5          T" X R                  5        SSS5        SSS5        g! , (       d  f       N= f! , (       d  f       g= f)a-  
DeepSpeed ZeRO-3 variant of `PreTrainedModel.initialize_weights`. Mirrors the `smart_apply`
dispatch logic but gathers each module's partitioned parameters before calling
`_initialize_weights`, so initialization operates on full tensors instead of empty shards.
Only rank 0 performs the actual init.
r   Nr   )Úguard_torch_init_functions)ÚPreTrainedModelc                 ó–  >• U R                  5        H0  n[        UT5      (       a  T" X"R                  5        M(  T" X!5        M2     [        U R	                  SS95      nU(       aK  TR
                  R                  USS9   TR                  R                  5       S:X  a	  U" U T5        S S S 5        g U" U T5        g ! , (       d  f       g = f)NF)Úrecurser   ©Úmodifier_rank)	ÚchildrenÚ
isinstanceÚ_initialize_weightsÚlistÚ
parametersÚzeroÚGatheredParametersÚcommÚget_rank)Úmodel_or_moduleÚfnÚchildÚparamsr‘   Ú_apply_zero3r   Úis_remote_codes       €€€€r   r£   Ú.initialize_weights_zero3.<locals>._apply_zero34  s«   ø€ Ø$×-Ñ-Ö/ˆEÜ˜% ×1Ñ1Ù˜U×$=Ñ$=Ö>á˜UÖ'ñ	 0ô �o×0Ñ0¸Ð0Ð?Ó@ˆÞØ—‘×2Ñ2°6ÈÐ2ÒKØ—>‘>×*Ñ*Ó,°Ó1Ù�¨Ô7÷ LÐKñ ˆ Õ/÷	 LÕKús   Á?(B:Â:
C)	r   rc   Úinitializationr�   Úmodeling_utilsr‘   r¤   Úno_gradr˜   )r|   rc   r�   r‘   r£   r   r¤   s      @@@@r   Úinitialize_weights_zero3r©   %  sc   û€ ó Ûå;Ý0à×)Ñ)Ó+€N÷0ð 0ð 
�Š�Ù'Õ)Ù˜× 9Ñ 9Ô:÷ *÷ 
ˆß)Õ)ú÷ 
�ús$   ÁB ÁA/ÁB Á/
A=	Á9B Â 
Bc                 óZ  ^!• [        5       nUb…  UR                  S0 5      R                  SS5      nUR                  S0 5      n[        U[        5      (       a+  [	        XER                  S0 5      R                  SS5      5      nUS:”  a  [        S5      eSS	KJnJnJ	m!J
n  [        US
S5      n	U R                  n
0 nU R                  5       R                  5        H1  u  pÍ[        R                   " UR"                  UR$                  SS9X¼'   M3     U Vs/ s H  n[        Xç5      (       d  M  UPM     nnU Vs/ s H  n[        Xæ5      (       d  M  UPM     nn['        U5      S:X  aC  0 nUR                  5        H!  u  nnU" UU/ X«5      u  nnUU;   d  M  UUU'   M#     U	b  U	Ul        U$ U VVs0 s H  nUR*                    H  nUU_M     M     nnn0 n0 n[-        UR/                  5       U!4S jS9nU H…  nUR1                  U5      nU" UUUX«5      u  nnUU;   d  M*  UbS  UU   nU" UR*                  UR2                  UR4                  S9nUR7                  UU5      nUR9                  UUUU5        M€  UUU'   M‡     UR                  5        H]  u  nn UR;                  UU U R<                  S9nUR                  5        H'  u  nn[        U[>        5      (       a  US   OUnUUU'   M)     M_     U	b  U	Ul        U$ s  snf s  snf s  snnf ! [@         a  n [C        SU SU  35      U eSn A ff = f)z°
Apply weight conversions (renaming and merging/splitting operations) to a state dict.
This is a simplified version that handles the conversion without loading into the model.
NÚtensor_parallelÚautotp_sizeé   Ú	inferenceÚtp_sizezóWeight conversions (e.g., MoE expert fusion) with DeepSpeed Tensor Parallelism are not yet implemented but support is coming soon. Please disable tensor_parallel in your DeepSpeed config or convert your checkpoint to the expected format first.r   )ÚWeightConverterÚWeightRenamingÚdot_natural_keyÚrename_source_keyÚ	_metadataÚmeta)r7   Údevicer   c                 ó   >• T" U 5      $ r1   r$   )Úkr²   s    €r   Ú<lambda>Ú9_apply_weight_conversions_to_state_dict.<locals>.<lambda>‚  s
   ø€ ¹/È!Ô:Lr#   )Úkey)Úsource_patternsÚtarget_patternsÚ
operations)r|   rH   z'Failed to apply weight conversion for 'zb'. This likely means the checkpoint format is incompatible with the current model version. Error: )"rŽ   rC   r—   Údictrw   ÚNotImplementedErrorÚcore_model_loadingr°   r±   r²   r³   ÚgetattrÚbase_model_prefixÚ
state_dictÚitemsrc   ÚemptyÚshaper7   ru   r´   r¼   ÚsortedÚkeysÚpopr½   r¾   Ú
setdefaultÚ
add_tensorÚconvertrH   r™   Ú	ExceptionÚRuntimeError)"r|   rÄ   Úweight_mappingÚ	ds_configr¯   Úinference_configr°   r±   r³   r   ÚprefixÚmodel_state_dictr»   ÚparamÚentryÚ	renamingsÚ
convertersÚnew_state_dictÚoriginal_keyÚtensorÚrenamed_keyr   Ú	converterr¸   Úpattern_to_converterÚconversion_mappingÚsorted_keysÚsource_patternÚnew_converterÚmappingÚrealized_valueÚtarget_nameÚer²   s"                                    @r   Ú'_apply_weight_conversions_to_state_dictrç   H  sg  ø€ ô !Ó"€IØÑà—-‘-Ð 1°2Ó6×:Ñ:¸=È!ÓLˆà$Ÿ=™=¨°bÓ9ÐÜÐ&¬×-Ñ-Ü˜'×#7Ñ#7Ð8IÈ2Ó#N×#RÑ#RÐS\Ð^_Ó#`ÓaˆGØ�Q‹;Ü%ðdóð ÷ iÓhô �z ;°Ó5€Hà×$Ñ$€Fð ÐØ×&Ñ&Ó(×.Ñ.Ö0‰
ˆÜ %§¢¨E¯K©K¸u¿{¹{ÐSYÑ ZÐÓñ 1ñ %3ÓX¢N˜5´jÀ×6W—¡N€IÐXÙ%3ÓZ¢^˜E´zÀ%×7Y—%¡^€JÐZô ˆ:ƒ˜!ÓØˆØ$.×$4Ñ$4Ö$6Ñ ˆL˜&Ù.¨|¸YÈÈFÓe‰NˆK˜ØÐ.Õ.Ø.4�˜{Ó+ñ %7ð
 ÑØ'/ˆNÔ$ØÐñ ;EÔhº*¨YÈi×NgÕNgÈ˜A˜yšLÑNg™A¹*ÐÑhð
 ÐØ€NÜ˜Ÿ™Ó*Ô0LÑM€KÛ#ˆØ—‘ Ó-ˆÙ&7¸ÀiÐQ[Ð]cÓ&vÑ#ˆ�^ð Ð*Õ*àÑ)ð 1°Ñ@�	Ù /Ø$-×$=Ñ$=Ø$-×$=Ñ$=Ø(×3Ñ3ñ!�ð
 -×7Ñ7¸À]ÓS�Ø×"Ñ" ;°¸nÈfÖUð /5�˜{Ó+ñ+ $ð0 !3× 8Ñ 8Ö :Ñˆ�Wð	Ø$Ÿ_™_ØØØ—|‘|ð -ð ˆNð
 '5×&:Ñ&:Ö&<Ñ"�˜UÜ$.¨u´d×$;Ñ$;˜˜ašÀ�Ø.3�˜{Ó+ó '=ñ !;ð$ ÑØ#+ˆÔ àÐùòK YùÚZùó iøôT ó 	ÜØ9¸+¸ð Gà˜ðóð ð	ûð	ús7   ÄK9Ä*K9Ä6K>ÅK>Æ-!LÊAL	Ì	
L*ÌL%Ì%L*c           	      ó  ^^	^
^• [        USS5      m
UR                  5       nT
b  T
Ul        SnUb  [        USS5      nUb!  [        U5      S:”  a  [	        XU5      nX0l        / mU R                  5       n[        UR                  5       5      m[        U SS5      nUR                  5        VVs0 s H&  u  pgUR                  U SU 35      b  U SU 3OUU_M(     nnnSS[        R                  4UU	U
U4S	 jjjm	T	" XSS
9  TT4$ s  snnf )a�  
Loads state dict into a model specifically for Zero3, since DeepSpeed does not support the `transformers`
tensor parallelism API.

Nearly identical code to PyTorch's `_load_from_state_dict`

Args:
    model_to_load: The model to load weights into
    state_dict: The state dict containing the weights
    load_config: Optional LoadStateDictConfig containing weight_mapping and other loading options
r´   NrÐ   r   rÃ   Ú.FÚmodulec                 ó¼  >• Tc  0 OTR                  US S 0 5      nX4S'   XUS/ / T4n[        5       (       GaL  SS Kn[        U R	                  US S SS95      n/ nU H7  n	X‘;   d  M
  Xy   n
SU
l        UR                  U
5        TR                  U	5        M9     [        U5      S:”  aT  UR                  R                  USS9   [        R                  R                  5       S:X  a  U R                  " U6   S S S 5        [        U R                  US S SS95      nUR!                  5        HZ  u  pœX‘;   d  M  Uc  M  TR                  U	5        [        R"                  " 5          UR%                  X   5        S S S 5        SUl        M\     U R&                  R!                  5        H  u  pÞUc  M
  T" XáX--   S-   U5        M     g ! , (       d  f       NÐ= f! , (       d  f       Nb= f)	NéÿÿÿÿÚassign_to_params_buffersTr   F)rÓ   r“   r”   ré   )rC   rŒ   r   r¿   Únamed_parametersÚ_is_hf_initializedrD   Údiscardru   r›   rœ   rc   Údistributedrž   Ú_load_from_state_dictÚnamed_buffersrÅ   r¨   Úcopy_Ú_modules)rê   rÄ   rÓ   rí   Úlocal_metadatarg   r   rî   Úparams_to_gatherr¸   rÕ   ró   ÚbufÚnamer¡   Ú
error_msgsÚloadr   Úmissing_keyss                  €€€€r   rû   Ú/_load_state_dict_into_zero3_model.<locals>.loadÝ  s¾  ø€ Ø'Ñ/™°X·\±\À&ÈÈ"À+ÈrÓ5RˆØ5MÐ1Ñ2à N°D¸"¸bÀ*ÐMˆô &×'Ò'Ûô  $ F×$;Ñ$;À6È#È2À;ÐX]Ð$;Ð$^Ó_ÐØ!ÐÛ%�Ø•?Ø,Ñ/�Eà/3�EÔ,Ø$×+Ñ+¨EÔ2Ø ×(Ñ(¨Ö+ñ &ô Ð#Ó$ qÓ(ð —^‘^×6Ñ6Ð7GÐWXÐ6ÒYÜ×(Ñ(×1Ñ1Ó3°qÓ8Ø×4Ò4°dÑ;÷ Zô
 ! ×!5Ñ!5¸VÀCÀR¸[ÐRWÐ!5Ð!XÓYˆMØ'×-Ñ-Ö/‘�Ø•? s£Ø ×(Ñ(¨Ô+ÜŸš�ØŸ	™	 *¡-Ô0÷ )à-1�CÖ*ñ 0ð "Ÿ?™?×0Ñ0Ö2‰KˆDØÓ Ù�U¨©¸Ñ(;Ð=UÖVò 3÷ ZÕYú÷ )�ús   Ã 2F<ÅGÆ<
G
Ç
G	)rí   )Ú F)rÂ   Úcopyr´   ru   rç   Ú_weight_conversionsrÄ   ÚsetrÉ   rÅ   rC   r	   ÚModule)Úmodel_to_loadrÄ   Úload_configrÐ   Úmeta_model_state_dictÚprefix_modelr¸   Úvrú   rû   r   rü   s           @@@@r   Ú!_load_state_dict_into_zero3_modelr  ³  s9  û€ ô �z ;°Ó5€HØ—‘Ó"€JØÑØ'ˆ
Ôð €NØÑÜ  Ð.>ÀÓEˆð Ñ!¤c¨.Ó&9¸AÓ&=Ü<¸]ÐXfÓgˆ
à,:Ô)à€JØ)×4Ñ4Ó6ÐÜÐ,×1Ñ1Ó3Ó4€Lä˜=Ð*=¸tÓD€Lð ×$Ñ$Ô&ôâ&‰DˆAð #8×";Ñ";¸|¸nÈAÈaÈSÐ<QÓ"RÑ"^ˆLˆ>˜˜1˜#Ñ	ÐdeÐhiÒ	iÙ&ð ñ ñ)W”R—Y‘Y÷ )Wó )WñV 	ˆ¸UÒCà�|Ð#Ð#ùóis   Â--Dc                 ó0  ^ ^• SSK JnJn  UR                  nSnSU;   a  U" US9nO?UR	                  5       (       a  [
        R                  S5        T R                  5       nSUS'   Sn	S	U;   a  U" U5      n	X‰4$ [        X…5      (       a  UU 4S
 jn
U" XŠS9n	X‰4$ )zQ
A convenience wrapper that deals with optimizer and lr scheduler configuration.
r   )Ú
DummyOptimÚDummySchedulerNÚ	optimizer)r¢   z¢Detected ZeRO Offload and non-DeepSpeed optimizers: This combination should work as long as the custom optimizer has both CPU and GPU implementation (except LAMB)TÚzero_allow_untested_optimizerÚ	schedulerc                 ób   >• [         R                   " T5      nS Ul        UR                  TU S9nU$ )N)r}   r  )rÿ   Úlr_schedulerÚcreate_scheduler)r  Útrainer_copyr  r}   Útrainers      €€r   Ú_lr_scheduler_callableÚ5deepspeed_optim_sched.<locals>._lr_scheduler_callable3  s=   ø€ ä#Ÿyšy¨Ó1�ð -1�Ô)Ø+×<Ñ<Ø'9ÀYð  =ð  �ð $Ð#r#   )Úlr_scheduler_callable)	Úaccelerate.utilsr
  r  rH   Ú
is_offloadÚloggerÚinfoÚcreate_optimizerr—   )r  Úhf_deepspeed_configrg   r}   Úmodel_parametersr
  r  rH   r  r  r  s   `  `       r   Údeepspeed_optim_schedr    s²   ù€ ÷ <à ×'Ñ'€Fð €IØ�fÓÙÐ&6Ñ7‰	à×)Ñ)×+Ñ+Ü�K‰KðVôð ×,Ñ,Ó.ˆ	à26ˆÐ.Ñ/à€LØ�fÓÙ% iÓ0ˆð" Ð"Ð"ô �i×,Ñ,ö	$ñ *¨)ÑbˆLàÐ"Ð"r#   c                 óÒ  • SSK Jn  U R                  nU R                  nU R                  R
                  R                  R                  nUR                  XTU5        UR                  UR                  5       5        U(       aK  UR                  5       (       d  [        S5      eUR                  S5        UR                  S5        Su  pxSn	Xx4$ SU l        UR                  R!                  S0 5      R!                  S	S
5      n
U
S
:”  a.  SSKnUR%                  UU
UR'                  5       UR                  S9n[)        [+        S UR-                  5       5      5      n	[/        XXQU	5      u  pxXx4$ )aÞ  
Init DeepSpeed, after updating the DeepSpeed configuration with any relevant Trainer's args.

If `resume_from_checkpoint` was passed then an attempt to resume from a previously saved checkpoint will be made.

Args:
    trainer: Trainer object
    num_training_steps: per single gpu
    resume_from_checkpoint: path to a checkpoint if to resume from after normal DeepSpeedEngine load
    inference: launch in inference mode (no optimizer and no lr scheduler)
    auto_find_batch_size: whether to ignore the `train_micro_batch_size_per_gpu` argument as it's being
        set automatically by the auto batch size finder

Returns: optimizer, lr_scheduler

We may use `deepspeed_init` more than once during the life of Trainer, when we do - it's a temp hack based on:
https://github.com/deepspeedai/DeepSpeed/issues/1394#issuecomment-937405374 until Deepspeed fixes a bug where it
can't resume from a checkpoint after it did some stepping https://github.com/deepspeedai/DeepSpeed/issues/1612

r   )r  zMZeRO inference only makes sense with ZeRO Stage 3 - please adjust your configr  r  )NNNr«   r¬   r­   )r|   r¯   r7   rH   c                 ó   • U R                   $ r1   )Úrequires_grad)Úps    r   r¹   Ú deepspeed_init.<locals>.<lambda>{  s   € °·²r#   )Údeepspeed.utilsr  r|   rg   ÚacceleratorÚstateÚdeepspeed_pluginÚhf_ds_configr�   ÚsetLevelÚget_process_log_levelrx   r6   Údel_config_sub_treer  rH   rC   r   Útp_model_initr7   r™   Úfilterrš   r  )r  r}   r®   Ú	ds_loggerr|   rg   r  r  r  r  Údeepspeed_tp_sizer   s               r   Údeepspeed_initr0  C  sd  € õ* 4à�M‰M€EØ�<‰<€Dà!×-Ñ-×3Ñ3×DÑD×QÑQÐð ×/Ñ/°Ð=OÔPð ×Ñ�t×1Ñ1Ó3Ô4æà"×+Ñ+×-Ñ-ÜÐlÓmÐmð 	×/Ñ/°Ô<Ø×/Ñ/°Ô?Ø",Ñˆ	ØÐð* Ð"Ð"ð' !ˆÔØ/×6Ñ6×:Ñ:Ð;LÈbÓQ×UÑUÐVcÐefÓgÐØ˜qÓ Ûà×+Ñ+ØØ)Ø)×/Ñ/Ó1Ø*×1Ñ1ð	 ,ð ˆEô  ¤Ñ'@À%×BRÑBRÓBTÓ UÓVÐÜ"7Ø¨$ÐDTó#
Ñˆ	ð Ð"Ð"r#   c                 óú   • SS K n[        UR                  U S35      5      n[        U5      S:”  a>  [        R	                  SU 35        U R                  UUSSS9u  pVUc  [        SU 35      eg [        SU 35      e)Nr   z/global_step*zAttempting to resume from T)Úload_module_strictÚload_optimizer_statesÚload_lr_scheduler_statesz-[deepspeed] failed to resume from checkpoint z!Can't find a valid checkpoint at )ÚglobrÈ   ru   r  r  Úload_checkpointr6   )Údeepspeed_engineÚcheckpoint_pathr2  r5  Údeepspeed_checkpoint_dirsÚ	load_pathr   s          r   Údeepspeed_load_checkpointr;  †  s    € ó
 ä & t§y¡y°OÐ3DÀMÐ1RÓ'SÓ TÐä
Ð$Ó%¨Ó)Ü�‰Ð0°Ð0AÐBÔCà'×7Ñ7ØØ1Ø"&Ø%)ð	 8ð 
‰ˆ	ð ÑÜÐLÈ_ÐL]Ð^Ó_Ð_ð ô Ð<¸_Ð<MÐNÓOÐOr#   c                 óä   • U R                   R                  n[        UR                  R                  5      Ul        UR                  R                  Ul        UR                  R                  X5        g)aw  
Sets values in the deepspeed plugin based on the TrainingArguments.

Args:
    accelerator (`Accelerator`): The Accelerator object.
    args (`TrainingArguments`): The training arguments to propagate to DeepSpeed config.
    auto_find_batch_size (`bool`, *optional*, defaults to `False`):
        Whether batch size was auto-discovered by trying increasingly smaller sizes.
N)r&  r'  r.   r(  rH   rŽ   ri   )r%  rg   rh   Ú	ds_plugins       r   Úpropagate_args_to_deepspeedr>  ž  sV   € ð ×!Ñ!×2Ñ2€Iä5°i×6LÑ6L×6SÑ6SÓT€IÔØ!*×!7Ñ!7×!>Ñ!>€IÔØ×Ñ×1Ñ1°$ÕMr#   c                 óà  ^^• SU;  a  SU;   a  US   US'   U" S0 UD6nUR                   nUR                  S:X  a'  UR                  S:”  a  SSKJn  UR                  5       nO6U R                  b  U R                  S   R                  5       nO[        S5      eUR                  n	[        R                  R                  R                  R                  XhS	9mUS   S
:g  R                  S5      R                  5       n
[        R                  R                  R                  R                  X¨S	9m[        UU4S j[!        U	5       5       5      n[        T5      nU[#        US5      -  nU(       a  Xe4$ U$ )aA  
Computes the loss under sequence parallelism with `sp_backend="deepspeed"` and `sp_size > 1`.

Performs weighted loss aggregation across SP ranks, accounting for varying numbers of valid tokens per rank
(e.g., when some ranks receive only padding or prompt tokens that are masked with -100).

Args:
    accelerator (`Accelerator`): The accelerator instance with `torch_device_mesh` support.
    model (`torch.nn.Module`): The model to compute the loss for.
    inputs (`dict[str, torch.Tensor | Any]`): The input data for the model. Must include `"shift_labels"` key.
    return_outputs (`bool`): Whether to return the model outputs along with the loss.
    pc (`accelerate.parallelism_config.ParallelismConfig`): The parallelism configuration.

Returns:
    The loss, or a tuple of `(loss, outputs)` if `return_outputs` is `True`.
ÚlabelsÚshift_labelsr   r­   r   )ÚgroupsÚspz™Sequence parallelism is enabled but no SP process group is available. Ensure torch_device_mesh is initialized or sp_backend='deepspeed' with sp_size > 1.)Úgroupiœÿÿÿrì   c              3   óP   >#   • U  H  nTU   S :”  d  M  TU   TU   -  v •  M     g7f)r   Nr$   )Ú.0ÚrankÚgood_tokens_per_rankÚlosses_per_ranks     €€r   Ú	<genexpr>Ú,deepspeed_sp_compute_loss.<locals>.<genexpr>Þ  s7   øé € ð â(ˆDØ Ñ%¨Ñ)ó 	;ˆ˜ÑÐ 4°TÑ :Ö:Ú(ùs   ƒ&”&r$   )ÚlossÚ
sp_backendÚsp_sizer$  rB  Ú_get_sequence_parallel_groupÚtorch_device_meshÚ	get_groupr6   rc   rñ   r	   Ú
functionalÚ
all_gatherÚviewÚsumÚrangerw   )r%  r|   ÚinputsÚreturn_outputsÚpcÚoutputsrL  rB  Úsp_groupÚsp_world_sizeÚgood_tokensÚ
total_lossÚtotal_good_tokensrH  rI  s                @@r   Údeepspeed_sp_compute_lossr`  ¯  s\  ù€ ð, �vÓ .°FÓ":à! .Ñ1ˆˆxÑÙ‰o�f‰o€GØ�<‰<€Dð 
‡}�}˜Ó#¨¯
©
°Q«Ý*à×6Ñ6Ó8‰Ø	×	&Ñ	&Ñ	2Ø×0Ñ0°Ñ6×@Ñ@ÓB‰äðbó
ð 	
ð —J‘J€Mä×'Ñ'×*Ñ*×5Ñ5×@Ñ@ÀÐ@ÐV€Oà˜.Ñ)¨TÑ1×7Ñ7¸Ó;×?Ñ?ÓA€KÜ ×,Ñ,×/Ñ/×:Ñ:×EÑEÀkÐEÐbÐäõ ä˜-Ô(óó €Jô
 Ð0Ó1ÐØœÐ-¨qÓ1Ñ1€Dæ,ˆDˆ?Ð6°$Ð6r#   r1   rƒ   )T)'r)   rÿ   Úimportlib.metadatar   Úimportlib.utilr…   Ú	functoolsr   Údependency_versions_checkr   Úutilsr   r   r   rc   r	   Ú
get_loggerr%   r  r   Úaccelerate.utils.deepspeedr   ÚDeepSpeedConfigÚbuiltinsr   r.   r‡   r   rŠ   rŒ   rŽ   r©   rç   r  r  r0  r;  r>  r`  r$   r#   r   Ú<module>rj     sÓ   ðñó Û Û Û Ý #å 9ß HÑ Hñ ×ÑÛÝð 
×	Ò	˜HÓ	%€ò
ñ ×ÑÑ!7×!9Ñ!9ÞOõ 3ô.˜ô .ô2pÐ0ô pðh !%Ð òIò)òòò ;òFhôVW$òt3#ôl@#ôFPô0Nó"77r#   