+
    QV-j†ƒ  ã                   ó„  € R t ^ RIt^ RIt^ RIt^ RIt^ RIHt ^RIH	t	 ^RI
HtHtHt ]! 4       '       d   ^ RIt^ RIHt ]P                   ! ]4      tR t]! 4       '       d   ]! 4       '       d   ^ RIHt M^ RIHt  ! R	 R
]4      t ! R R]4      tRsR tR tR tR tR tR t RR lt!R t"RR lt#RR lt$RR lt%R t&R# )z
Integration with Deepspeed
N)Úpartialmethod)Údep_version_check)Úis_accelerate_availableÚis_torch_availableÚlogging)Únnc                  óè   € \         P                  P                  R 4      RJp V '       d#    \         P                  P                  R 4      pR# R#   \         P                  P                   d     R# i ; i)Ú	deepspeedNTF)Ú	importlibÚutilÚ	find_specÚmetadataÚPackageNotFoundError)Úpackage_existsÚ_s     Út/Volumes/fast/ai/experiments/ui-tars-smoke/.venv/lib/python3.14/site-packages/transformers/integrations/deepspeed.pyÚis_deepspeed_availabler   $   sc   € Ü—^‘^×-Ñ-¨kÓ:À$ÐF€N÷ ð	Ü×"Ñ"×+Ñ+¨KÓ8ˆAÙñ øô ×!Ñ!×6Ñ6ô 	Úð	ús   «A ÁA1Á0A1)ÚHfDeepSpeedConfig)Úobjectc                   ó6   a a€ ] tR t^9t oRtV 3R ltRtVtV ;t# )r   a"  
This object contains a DeepSpeed configuration dictionary and can be quickly queried for things like zero stage.

A `weakref` of this object is stored in the module's globals to be able to access the config from areas where
things like the Trainer object is not available (e.g. `from_pretrained` and `_get_resized_embeddings`). Therefore
it's important that this object remains alive while the program is still running.

[`Trainer`] uses the `HfTrainerDeepSpeedConfig` subclass instead. That subclass has logic to sync the configuration
with values of [`TrainingArguments`] by replacing special placeholder values: `"auto"`. Without this special logic
the DeepSpeed configuration is not modified in any way.

Args:
    config_file_or_dict (`Union[str, Dict]`): path to DeepSpeed config file or dict.

c                óh   <€ \        V 4       \        R 4       \        R4       \        SV `  V4       R# )Ú
accelerater	   N)Úset_hf_deepspeed_configr   ÚsuperÚ__init__©ÚselfÚconfig_file_or_dictÚ	__class__s   &&€r   r   ÚHfDeepSpeedConfig.__init__J   s)   ø€ ä Ô%Ü˜,Ô'Ü˜+Ô&Ü‰ÑÐ,Ö-ó    © )	Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__r   Ú__static_attributes__Ú__classdictcell__Ú__classcell__©r   Ú__classdict__s   @@r   r   r   9   s   ù‡ € ñ÷ .õ .r    r   c                   óp   a a€ ] tR t^Rt oRtV 3R ltR tR tRR lt]	! ]RR7      t
RR ltR	 tR
tVtV ;t# )ÚHfTrainerDeepSpeedConfigz’
The `HfTrainerDeepSpeedConfig` object is meant to be created during `TrainingArguments` object creation and has the
same lifespan as the latter.
c                óB   <€ \         SV `  V4       R V n        . V n        R # ©N)r   r   Ú_dtypeÚ
mismatchesr   s   &&€r   r   Ú!HfTrainerDeepSpeedConfig.__init__X   s   ø€ Ü‰ÑÐ,Ô-ØˆŒØˆŽr    c                óL   € V P                   f   \        R4      hV P                   # )Nz8trainer_config_process() wasn't called yet to tell dtype)r0   Ú
ValueError)r   s   &r   ÚdtypeÚHfTrainerDeepSpeedConfig.dtype]   s"   € Ø�;‰;ÒÜÐWÓXÐXØ�{‰{Ðr    c                ó:   € V P                  V4      pVf   R# VR8H  # )NFÚauto)Ú	get_value)r   Úds_key_longÚvals   && r   Úis_autoÚ HfTrainerDeepSpeedConfig.is_autob   s"   € Ø�n‰n˜[Ó)ˆØŠ;Ùà˜&‘=Ð r    c           
     ó  € V P                  V4      w  rVVf   R# VP                  V4      R8X  d   W%V&   R# V'       g   R# VP                  V4      pVe2   Wr8w  d*   V P                  P                  RV RV RV RV 24       R# R# R# )a†  
A utility method that massages the config file and can optionally verify that the values match.

1. Replace "auto" values with `TrainingArguments` value.

2. If it wasn't "auto" and `must_match` is true, then check that DS config matches Trainer
config values and if mismatched add the entry to `self.mismatched` - will assert during
`trainer_config_finalize` for one or more mismatches.

Nr8   z- ds Ú=z vs hf )Úfind_config_nodeÚgetr1   Úappend)r   r:   Úhf_valÚhf_keyÚ
must_matchÚconfigÚds_keyÚds_vals   &&&&&   r   Ú
fill_matchÚ#HfTrainerDeepSpeedConfig.fill_matchi   s�   € ð ×.Ñ.¨{Ó;‰ˆØŠ>Ùà�:‰:�fÓ Ô'Ø#�6‰NÙçÙà—‘˜FÓ#ˆØÒ &Ô"2Ø�O‰O×"Ñ" U¨;¨-°q¸¸ÀÈÀxÈqÐQWÐPXÐ#YÖZñ #3Ñr    F)rE   c                ó  € VP                   VP                  ,          VP                  ,          pV P                  RVP                  RV'       * 4       V P                  RVP                  R4       V P                  RVRV'       * 4       V P                  RVP                  R4       V P                  RVP
                  R	4       V P                  R
VP                  VP                  .R4       V P                  RVP                  R4       V P                  RVP                  R4       V P                  R^ 4       V P                  RVP
                  R	4       VP                  '       dJ   V P                  P                  R/ 4      V P                  R&   VP                  V P                  R,          R&   T P                  RVP                  ;'       g    VP                  R4       T P                  RVP                   ;'       g    VP"                  R4       V P%                  R4      '       d   \&        P(                  V n        R# V P%                  R4      '       d   \&        P,                  V n        R# \&        P.                  V n        R# )zr
Adjust the config with `TrainingArguments` values. This stage is run during `TrainingArguments` object
creation.
Útrain_micro_batch_size_per_gpuÚper_device_train_batch_sizeÚgradient_accumulation_stepsÚtrain_batch_sizeztrain_batch_size (calculated)Úgradient_clippingÚmax_grad_normzoptimizer.params.lrÚlearning_ratezoptimizer.params.betaszadam_beta1+adam_beta2zoptimizer.params.epsÚadam_epsilonzoptimizer.params.weight_decayÚweight_decayzscheduler.params.warmup_min_lrzscheduler.params.warmup_max_lrÚ
checkpointÚuse_node_local_storagezfp16.enabledzfp16|fp16_full_evalzbf16.enabledzbf16|bf16_full_evalN)Ú
world_sizerM   rN   rI   rQ   rR   Ú
adam_beta1Ú
adam_beta2rS   rT   Ú	fill_onlyÚsave_on_each_noderF   rA   Úfp16Úfp16_full_evalÚbf16Úbf16_full_evalÚis_trueÚtorchÚbfloat16r0   Úfloat16Úfloat32)r   ÚargsÚauto_find_batch_sizerO   s   &&& r   Útrainer_config_processÚ/HfTrainerDeepSpeedConfig.trainer_config_process…   sö  € ð  Ÿ?™?¨T×-MÑ-MÕMÐPT×PpÑPpÕpÐØ�‰Ø,Ø×,Ñ,Ø)Ø$Ô$ô		
ð 	�‰Ø)Ø×,Ñ,Ø)ô	
ð
 	�‰ØØØ+Ø$Ô$ô		
ð 	�‰Ð+¨T×-?Ñ-?ÀÔQà�‰Ð-¨t×/AÑ/AÀ?ÔSØ�‰Ø$Ø�_‰_˜dŸo™oÐ.Ø#ô	
ð
 	�‰Ð.°×0AÑ0AÀ>ÔRØ�‰Ð7¸×9JÑ9JÈNÔ[à�‰Ð7¸Ô;Ø�‰Ð8¸$×:LÑ:LÈoÔ^ð ×!×!Ð!à(,¯©¯©¸ÀbÓ(IˆD�K‰K˜Ñ%ØBF×BXÑBXˆD�K‰K˜Õ%Ð&>Ñ?ð 	�‰˜¨¯©×)IÐ)I°d×6IÑ6IÐLaÔbØ�‰˜¨¯©×)IÐ)I°d×6IÑ6IÐLaÔbð �<‰<˜×'Ò'ÜŸ.™.ˆDŽKØ�\‰\˜.×)Ò)ÜŸ-™-ˆDŽKäŸ-™-ˆDŽKr    c                ó*  € . ROpV Uu. uF  qPP                  V4      '       g   K  VNK  	  pp\        V4      ^ 8”  Ed×   Rp\        VR4      '       Ed?   \        VP                  R4      '       d   VP                  P                  pEM
\        VP                  R4      '       d!   \        VP                  P                  4      pMÎ\        VP                  R4      '       dH   \        VP                  P                  R4      '       d"   VP                  P                  P                  pMk\        VP                  R4      '       dP   \        VP                  P                  R4      '       d*   \        VP                  P                  P                  4      pVf   \        R	V R
24      hV P                  RWw,          4       V P                  4       '       dC   V P                  R\        RV,          V,          4      4       V P                  R^
V,          4       V P                  RVR4       V P                  RVP                  V4      R4       \        V P                  4      ^ 8”  d+   RP                  V P                  4      p\        RV R24      hR# u upi )zx
This stage is run after we have the model and know num_training_steps.

Now we can complete the configuration process.
ú$zero_optimization.reduce_bucket_sizeú-zero_optimization.stage3_prefetch_bucket_sizeú4zero_optimization.stage3_param_persistence_thresholdNrF   Úhidden_sizeÚhidden_sizesÚtext_configz½The model's config file has neither `hidden_size` nor `hidden_sizes` entry, therefore it's not possible to automatically fill out the following `auto` entries in the DeepSpeed config file: zb. You can fix that by replacing `auto` values for these keys with an integer value of your choice.gÍÌÌÌÌÌì?z scheduler.params.total_num_stepsznum_training_steps (calculated)z!scheduler.params.warmup_num_stepsÚwarmup_stepsÚ
z]Please correct the following DeepSpeed config values that mismatch TrainingArguments values:
zF
The easiest method is to set these DeepSpeed config values to 'auto'.)rj   rk   rl   )r<   ÚlenÚhasattrrF   rm   Úmaxrn   ro   r4   rZ   Úis_zero3ÚintrI   Úget_warmup_stepsr1   Újoin)	r   re   ÚmodelÚnum_training_stepsÚhidden_size_based_keysÚxÚhidden_size_auto_keysrm   r1   s	   &&&&     r   Útrainer_config_finalizeÚ0HfTrainerDeepSpeedConfig.trainer_config_finalize¿   s"  € ò"
Ðñ
 -CÓ VÑ,B qÇlÁlÐSTÇo§ Ñ,BÐÐ VäÐ$Ó%¨Õ)ØˆKÜ�u˜h×'Ó'Ü˜5Ÿ<™<¨×7Ò7Ø"'§,¡,×":Ñ":’KÜ˜UŸ\™\¨>×:Ò:ä"% e§l¡l×&?Ñ&?Ó"@‘KÜ˜UŸ\™\¨=×9Ò9¼gÀeÇlÁl×F^ÑF^Ð`m×>nÒ>nØ"'§,¡,×":Ñ":×"FÑ"F‘KÜ˜UŸ\™\¨=×9Ò9¼gÀeÇlÁl×F^ÑF^Ð`n×>oÒ>oä"% e§l¡l×&>Ñ&>×&KÑ&KÓ"L�KàÒ"Ü ð5à5JÐ4Kð LYðYóð ð �N‰NÐAÀ;ÕC\Ô]Ø�}‰}�Šà—‘ØCÜ˜˜kÕ)¨KÕ7Ó8ôð —‘ØJØ˜Õ$ôð 	�‰Ø.ØØ-ô	
ð
 	�‰Ø/Ø×!Ñ!Ð"4Ó5Øô	
ô ˆt�‰Ó !Ô#ØŸ™ 4§?¡?Ó3ˆJÜðØ'˜LÐ(oðqóð ñ $ùòa !Ws
   ‰J¦J)r0   r1   )NT©F)r"   r#   r$   r%   r&   r   r5   r<   rI   r   rZ   rg   r~   r'   r(   r)   r*   s   @@r   r-   r-   R   s?   ù‡ € ñõ
ò
ò
!ô[ñ4 ˜j°UÔ;€Iô8(÷tCò Cr    r-   c                 ó2   € \         P                  ! V 4      sR # r/   )ÚweakrefÚrefÚ_hf_deepspeed_config_weak_ref)Úhf_deepspeed_config_objs   &r   r   r   	  s   € ô
 %,§K¢KÐ0GÓ$HÒ!r    c                  ó
   € R s R # r/   )r„   r!   r    r   Úunset_hf_deepspeed_configr‡     s
   € ð %)Ò!r    c                  ó^   € \         e%   \        4       e   \        4       P                  4       # R# )NF)r„   ru   r!   r    r   Úis_deepspeed_zero3_enabledr‰     s&   € Ü$Ò0Ô5RÓ5TÒ5`Ü,Ó.×7Ñ7Ó9Ð9ár    c                  óV   € \         e!   \        4       e   \        4       P                  # R # r/   )r„   rF   r!   r    r   Údeepspeed_configr‹     s#   € Ü$Ò0Ô5RÓ5TÒ5`Ü,Ó.×5Ñ5Ð5ár    c           	     óN  aaaa€ ^ RI o^ RIp^RIHp ^RIHo V P                  4       oVVVV3R loVP                  ! 4       ;_uu_ 4        V! 4       ;_uu_ 4        S! W P                  4       RRR4       RRR4       R#   + '       g   i     L; i  + '       g   i     R# ; i)a-  
DeepSpeed ZeRO-3 variant of `PreTrainedModel.initialize_weights`. Mirrors the `smart_apply`
dispatch logic but gathers each module's partitioned parameters before calling
`_initialize_weights`, so initialization operates on full tensors instead of empty shards.
Only rank 0 performs the actual init.
N)Úguard_torch_init_functions)ÚPreTrainedModelc                 óÂ  <€ V P                  4        F1  p\        VS4      '       d   S! W"P                  4       K)  S! W!4       K3  	  \        V P	                  R R7      4      pV'       dY   SP
                  P                  V^ R7      ;_uu_ 4        SP                  P                  4       ^ 8X  d
   V! V S4       RRR4       R# V! V S4       R#   + '       g   i     R# ; i)F)Úrecurse©Úmodifier_rankN)	ÚchildrenÚ
isinstanceÚ_initialize_weightsÚlistÚ
parametersÚzeroÚGatheredParametersÚcommÚget_rank)Úmodel_or_moduleÚfnÚchildÚparamsrŽ   Ú_apply_zero3r	   Úis_remote_codes   &&  €€€€r   r    Ú.initialize_weights_zero3.<locals>._apply_zero34  s±   ø€ Ø$×-Ñ-Ö/ˆEÜ˜% ×1Ò1Ù˜U×$=Ñ$=Ö>á˜UÖ'ñ	 0ô �o×0Ñ0¸Ð0Ó?Ó@ˆßØ—‘×2Ñ2°6ÈÐ2×KÕKØ—>‘>×*Ñ*Ó,°Ô1Ù�¨Ô7÷ LÑKñ ˆ Ö/÷	 L×KÐKús   Â)CÃC	)	r	   ra   Úinitializationr�   Úmodeling_utilsrŽ   r¡   Úno_gradr•   )ry   ra   r�   rŽ   r    r	   r¡   s   &  @@@@r   Úinitialize_weights_zero3r¦   %  sj   û€ ó Ûå;Ý0à×)Ñ)Ó+€N÷0ð 0ð 
�Š��Ù'×)Õ)Ù˜× 9Ñ 9Ô:÷ *÷ 
‰ß)×)ú÷ 
�ˆús$   ÁBÁB 	Á.BÂ BÂBÂB$	c           	     ó˜  a!€ \        4       pVeˆ   VP                  R/ 4      P                  R^4      pVP                  R/ 4      p\        V\        4      '       d,   \	        WEP                  R/ 4      P                  R^4      4      pV^8”  d   \        R4      h^RIHpHpH	o!H
p \        VRR4      p	V P                  p
/ pV P                  4       P                  4        F4  w  rÍ\        P                   ! VP"                  VP$                  R	R
7      W¼&   K6  	  V Uu. uF  p\        Wç4      '       g   K  VNK  	  ppV Uu. uF  p\        Wæ4      '       g   K  VNK  	  pp\'        V4      ^ 8X  dG   / pVP                  4        F#  w  ppV! VV. W«R7      w  ppVV9   g   K  VVV&   K%  	  V	e   V	Vn        V# V UUu/ uF  pVP*                   F  pVVbK  	  K  	  ppp/ p/ p\-        VP/                  4       V!3R lR7      pV F�  pVP1                  V4      pV! VVVW«R7      w  ppVV9   g   K,  Ve[   VV,          pV! VP*                  VP2                  VP4                  R7      pVP7                  VV4      pVP9                  VVVV4       KŠ  VVV&   K‘  	  VP                  4        Fe  w  pp VP;                  VV V P<                  R7      pVP                  4        F,  w  pp\        V\>        4      '       d
   V^ ,          MTpVVV&   K.  	  Kg  	  V	e   V	Vn        V# u upi u upi u uppi   \@         d   p \C        RT RT  24      T hRp ? ii ; i)z°
Apply weight conversions (renaming and merging/splitting operations) to a state dict.
This is a simplified version that handles the conversion without loading into the model.
NÚtensor_parallelÚautotp_sizeÚ	inferenceÚtp_sizezóWeight conversions (e.g., MoE expert fusion) with DeepSpeed Tensor Parallelism are not yet implemented but support is coming soon. Please disable tensor_parallel in your DeepSpeed config or convert your checkpoint to the expected format first.)ÚWeightConverterÚWeightRenamingÚdot_natural_keyÚrename_source_keyÚ	_metadataÚmeta)r5   Údevice)Úbase_model_prefixÚmeta_state_dictc                 ó   <€ S! V 4      # r/   r!   )Úkr®   s   &€r   Ú<lambda>Ú9_apply_weight_conversions_to_state_dict.<locals>.<lambda>„  s
   ø€ ¹/È!Ô:Lr    )Úkey)Úsource_patternsÚtarget_patternsÚ
operations)ry   rF   z'Failed to apply weight conversion for 'zb'. This likely means the checkpoint format is incompatible with the current model version. Error: )"r‹   rA   r”   Údictrt   ÚNotImplementedErrorÚcore_model_loadingr¬   r­   r®   r¯   Úgetattrr³   Ú
state_dictÚitemsra   ÚemptyÚshaper5   rr   r°   rº   ÚsortedÚkeysÚpopr»   r¼   Ú
setdefaultÚ
add_tensorÚconvertrF   r–   Ú	ExceptionÚRuntimeError)"ry   rÁ   Úweight_mappingÚ	ds_configr«   Úinference_configr¬   r­   r¯   r   r³   Úmodel_state_dictr¹   ÚparamÚentryÚ	renamingsÚ
convertersÚnew_state_dictÚoriginal_keyÚtensorÚrenamed_keyr   Ú	converterr¶   Úpattern_to_converterÚconversion_mappingÚsorted_keysÚsource_patternÚnew_converterÚmappingÚrealized_valueÚtarget_nameÚer®   s"   &&&                              @r   Ú'_apply_weight_conversions_to_state_dictrã   H  st  ø€ ô !Ó"€IØÒà—-‘-Ð 1°2Ó6×:Ñ:¸=È!ÓLˆà$Ÿ=™=¨°bÓ9ÐÜÐ&¬×-Ò-Ü˜'×#7Ñ#7Ð8IÈ2Ó#N×#RÑ#RÐS\Ð^_Ó#`ÓaˆGØ�QŒ;Ü%ðdóð ÷ iÓhô �z ;°Ó5€Hà×/Ñ/Ðð ÐØ×&Ñ&Ó(×.Ñ.Ö0‰
ˆÜ %§¢¨E¯K©K¸u¿{¹{ÐSYÔ ZÐÓñ 1ñ %3ÓX¡N˜5´jÀ×6W—�¡N€IÐXÙ%3ÓZ¡^˜E´zÀ%×7Y—%�%¡^€JÐZô ˆ:ƒ˜!ÔØˆØ$.×$4Ñ$4Ö$6Ñ ˆL˜&Ù.Ø˜i¨Ð?Pô‰NˆK˜ð Ð.Ö.Ø.4�˜{Ó+ñ %7ð ÒØ'/ˆNÔ$ØÐñ ;EÔh¹*¨YÈi×NgÔNgÈ˜A˜yšLÑNg™A¹*ÐÑhð
 ÐØ€NÜ˜Ÿ™Ó*Ô0LÔM€KÛ#ˆØ—‘ Ó-ˆÙ&7Ø˜) ZÐCTô'
Ñ#ˆ�^ð
 Ð*Ö*àÒ)ð 1°Õ@�	Ù /Ø$-×$=Ñ$=Ø$-×$=Ñ$=Ø(×3Ñ3ô!�ð
 -×7Ñ7¸À]ÓS�Ø×"Ñ" ;°¸nÈfÖUð /5�˜{Ó+ñ/ $ð4 !3× 8Ñ 8Ö :Ñˆ�Wð	Ø$Ÿ_™_ØØØ—|‘|ð -ó ˆNð
 '5×&:Ñ&:Ö&<Ñ"�˜UÜ$.¨u´d×$;Ò$;˜˜ažÀ�Ø.3�˜{Ó+ó '=ñ !;ð$ ÒØ#+ˆÔ àÐùòS YùÚZùó" iøôX ô 	ÜØ9¸+¸ð Gà˜ðóð ð	ûð	ús7   ÄLÄ0LÄ<LÅLÆ7L!Ê(AL'Ì'M	Ì2MÍM	c           	     ó  aa	a
a€ \        VRR4      o
VP                  4       pS
e   S
Vn        RpVe   \        VRR4      pVe#   \        V4      ^ 8”  d   \	        WV4      pW0n        . oV P                  4       p\        VP                  4       4      o\        V RR4      pVP                  4        UUu/ uF'  w  rgVP                  V RV 24      e   V RV 2MTVbK)  	  pppR
R VV	V
V3R lllo	S	! WRR	7       SS3# u uppi )a�  
Loads state dict into a model specifically for Zero3, since DeepSpeed does not support the `transformers`
tensor parallelism API.

Nearly identical code to PyTorch's `_load_from_state_dict`

Args:
    model_to_load: The model to load weights into
    state_dict: The state dict containing the weights
    load_config: Optional LoadStateDictConfig containing weight_mapping and other loading options
r°   NrÍ   r³   Ú.Fc                ó8   € V ^8„  d   QhR\         P                  /# )é   Úmodule)r   ÚModule)Úformats   "r   Ú__annotate__Ú7_load_state_dict_into_zero3_model.<locals>.__annotate__á  s   € ÷ )Wñ )W”R—Y‘Yñ )Wr    c                 ó*  <€ Sf   / MSP                  VR R / 4      pW4R&   WVR. . S3p\        4       '       Edt   ^ R Ip\        V P	                  VR R RR7      4      p. pV F<  p	W‘9   g   K  Wy,          p
RV
n        VP                  V
4       SP                  V	4       K>  	  \        V4      ^ 8”  db   VP                  P                  V^ R7      ;_uu_ 4        \        P                  P                  4       ^ 8X  d   V P                  ! V!   R R R 4       \        V P                  VR R RR7      4      pVP!                  4        Fh  w  rœW‘9   g   K  Vf   K  SP                  V	4       \        P"                  ! 4       ;_uu_ 4        VP%                  W,          4       R R R 4       RVn        Kj  	  V P&                  P!                  4        F"  w  rÞVf   K  S! WáW-,           R,           V4       K$  	  R #   + '       g   i     Lí; i  + '       g   i     Lp; i)NÚassign_to_params_buffersTF)Úprefixr�   r‘   rå   éÿÿÿÿ)rA   r‰   r	   r½   Únamed_parametersÚ_is_hf_initializedrB   Údiscardrr   r˜   r™   ra   Údistributedr›   Ú_load_from_state_dictÚnamed_buffersrÂ   r¥   Úcopy_Ú_modules)rè   rÁ   rï   rî   Úlocal_metadatare   r	   rñ   Úparams_to_gatherr¶   rÑ   rö   ÚbufÚnamerž   Ú
error_msgsÚloadr   Úmissing_keyss   &&&&           €€€€r   rþ   Ú/_load_state_dict_into_zero3_model.<locals>.loadá  sÃ  ø€ Ø'Ò/™°X·\±\À&ÈÈ"À+ÈrÓ5RˆØ5MÐ1Ñ2à N°D¸"¸bÀ*ÐMˆô &×'Ó'Ûô  $ F×$;Ñ$;À6È#È2À;ÐX]Ð$;Ó$^Ó_ÐØ!ÐÛ%�Ø–?Ø,Õ/�Eà/3�EÔ,Ø$×+Ñ+¨EÔ2Ø ×(Ñ(¨Ö+ñ &ô Ð#Ó$ qÔ(ð —^‘^×6Ñ6Ð7GÐWXÐ6×YÕYÜ×(Ñ(×1Ñ1Ó3°qÔ8Ø×4Ò4°dÒ;÷ Zô
 ! ×!5Ñ!5¸VÀCÀR¸[ÐRWÐ!5Ó!XÓYˆMØ'×-Ñ-Ö/‘�Ø–? s¤Ø ×(Ñ(¨Ô+ÜŸšŸ�ØŸ	™	 *¥-Ô0÷ )à-1�CÖ*ñ 0ð "Ÿ?™?×0Ñ0Ö2‰KˆDØÔ Ù�U¨­¸Õ(;Ð=UÖVó 3÷ Z×Yú÷ )Ÿús   Ã4G/ÆHÇ/G?	ÈH)rî   )Ú F)rÀ   Úcopyr°   rr   rã   Ú_weight_conversionsrÁ   ÚsetrÆ   rÂ   rA   )Úmodel_to_loadrÁ   Úload_configrÍ   Úmeta_model_state_dictÚprefix_modelr¶   Úvrý   rþ   r   rÿ   s   &&&     @@@@r   Ú!_load_state_dict_into_zero3_modelr
  ·  s-  û€ ô �z ;°Ó5€HØ—‘Ó"€JØÒØ'ˆ
Ôð €NØÒÜ  Ð.>ÀÓEˆð Ò!¤c¨.Ó&9¸AÔ&=Ü<¸]ÐXfÓgˆ
à,:Ô)à€JØ)×4Ñ4Ó6ÐÜÐ,×1Ñ1Ó3Ó4€Lä˜=Ð*=¸tÓD€Lð ×$Ñ$Ô&ôá&‰DˆAð #8×";Ñ";¸|¸nÈAÈaÈSÐ<QÓ"RÒ"^ˆLˆ>˜˜1˜#Ñ	ÐdeÐhiÒ	iÙ&ð ñ ÷)Wõ )WñV 	ˆ¸UÕCà�|Ð#Ð#ùóis   Â1-C=c                óD  a a€ ^ RI HpHp VP                  pRpRV9   d   V! VR7      pM@VP	                  4       '       d   \
        P                  R4       S P                  4       pRVR&   Rp	RV9   d   V! V4      p	W‰3# \        W…4      '       d   VV 3R	 lp
V! WŠR
7      p	W‰3# )zQ
A convenience wrapper that deals with optimizer and lr scheduler configuration.
)Ú
DummyOptimÚDummySchedulerNÚ	optimizer)rŸ   z¢Detected ZeRO Offload and non-DeepSpeed optimizers: This combination should work as long as the custom optimizer has both CPU and GPU implementation (except LAMB)TÚzero_allow_untested_optimizerÚ	schedulerc                 óh   <€ \         P                   ! S4      pR Vn        VP                  SV R7      pV# )N)rz   r  )r  Úlr_schedulerÚcreate_scheduler)r  Útrainer_copyr  rz   Útrainers   &  €€r   Ú_lr_scheduler_callableÚ5deepspeed_optim_sched.<locals>._lr_scheduler_callable7  s=   ø€ ä#Ÿyšy¨Ó1�ð -1�Ô)Ø+×<Ñ<Ø'9ÀYð  =ó  �ð $Ð#r    )Úlr_scheduler_callable)	Úaccelerate.utilsr  r  rF   Ú
is_offloadÚloggerÚinfoÚcreate_optimizerr”   )r  Úhf_deepspeed_configre   rz   Úmodel_parametersr  r  rF   r  r  r  s   f&&f&      r   Údeepspeed_optim_schedr     s²   ù€ ÷ <à ×'Ñ'€Fð €IØ�fÔÙÐ&6Ô7‰	à×)Ñ)×+Ò+Ü�K‰KðVôð ×,Ñ,Ó.ˆ	à26ˆÐ.Ñ/à€LØ�fÔÙ% iÓ0ˆð" Ð"Ð"ô �i×,Ò,ö	$ñ *¨)ÔbˆLàÐ"Ð"r    c                óÜ  € ^ RI Hp V P                  pV P                  pV P                  P
                  P                  P                  pVP                  WTV4       VP                  VP                  4       4       V'       dL   VP                  4       '       g   \        R4      hVP                  R4       VP                  R4       RRr‡Rp	Wx3# RV n        VP                  P!                  R/ 4      P!                  R^4      p
V
^8”  d2   ^ RIpVP%                  VV
VP'                  4       VP                  R7      p\)        \+        R	 VP-                  4       4      4      p	\/        WWQV	4      w  rxWx3# )
aÞ  
Init DeepSpeed, after updating the DeepSpeed configuration with any relevant Trainer's args.

If `resume_from_checkpoint` was passed then an attempt to resume from a previously saved checkpoint will be made.

Args:
    trainer: Trainer object
    num_training_steps: per single gpu
    resume_from_checkpoint: path to a checkpoint if to resume from after normal DeepSpeedEngine load
    inference: launch in inference mode (no optimizer and no lr scheduler)
    auto_find_batch_size: whether to ignore the `train_micro_batch_size_per_gpu` argument as it's being
        set automatically by the auto batch size finder

Returns: optimizer, lr_scheduler

We may use `deepspeed_init` more than once during the life of Trainer, when we do - it's a temp hack based on:
https://github.com/deepspeedai/DeepSpeed/issues/1394#issuecomment-937405374 until Deepspeed fixes a bug where it
can't resume from a checkpoint after it did some stepping https://github.com/deepspeedai/DeepSpeed/issues/1612

)r  zMZeRO inference only makes sense with ZeRO Stage 3 - please adjust your configr  r  Nr¨   r©   )ry   r«   r5   rF   c                 ó   € V P                   # r/   )Úrequires_grad)Úps   &r   r·   Ú deepspeed_init.<locals>.<lambda>  s   € °·²r    )Údeepspeed.utilsr  ry   re   ÚacceleratorÚstateÚdeepspeed_pluginÚhf_ds_configr~   ÚsetLevelÚget_process_log_levelru   r4   Údel_config_sub_treer  rF   rA   r	   Útp_model_initr5   r–   Úfilterr—   r   )r  rz   rª   Ú	ds_loggerry   re   r  r  r  r  Údeepspeed_tp_sizer	   s   &&&         r   Údeepspeed_initr2  G  sc  € õ* 4à�M‰M€EØ�<‰<€Dà!×-Ñ-×3Ñ3×DÑD×QÑQÐð ×/Ñ/°Ð=OÔPð ×Ñ�t×1Ñ1Ó3Ô4çà"×+Ñ+×-Ò-ÜÐlÓmÐmð 	×/Ñ/°Ô<Ø×/Ñ/°Ô?Ø"&¨�<ØÐð* Ð"Ð"ð' !ˆÔØ/×6Ñ6×:Ñ:Ð;LÈbÓQ×UÑUÐVcÐefÓgÐØ˜qÔ Ûà×+Ñ+ØØ)Ø)×/Ñ/Ó1Ø*×1Ñ1ð	 ,ó ˆEô  ¤Ñ'@À%×BRÑBRÓBTÓ UÓVÐÜ"7Ø¨$ÐDTó#
Ñˆ	ð Ð"Ð"r    c                 ó  € ^ RI p\        VP                  V R24      4      p\        V4      ^ 8”  dD   \        P	                  RV 24       V P                  VVRRR7      w  rVVf   \        RV 24      hR# \        RV 24      h)é    Nz/global_step*zAttempting to resume from T)Úload_module_strictÚload_optimizer_statesÚload_lr_scheduler_statesz-[deepspeed] failed to resume from checkpoint z!Can't find a valid checkpoint at )ÚglobrÅ   rr   r  r  Úload_checkpointr4   )Údeepspeed_engineÚcheckpoint_pathr5  r8  Údeepspeed_checkpoint_dirsÚ	load_pathr   s   &&&    r   Údeepspeed_load_checkpointr>  Š  s    € ó
 ä & t§y¡y°OÐ3DÀMÐ1RÓ'SÓ TÐä
Ð$Ó%¨Ô)Ü�‰Ð0°Ð0AÐBÔCà'×7Ñ7ØØ1Ø"&Ø%)ð	 8ó 
‰ˆ	ð ÒÜÐLÈ_ÐL]Ð^Ó_Ð_ñ ô Ð<¸_Ð<MÐNÓOÐOr    c                óæ   € V P                   P                  p\        VP                  P                  4      Vn        VP                  P                  Vn        VP                  P                  W4       R# )aw  
Sets values in the deepspeed plugin based on the TrainingArguments.

Args:
    accelerator (`Accelerator`): The Accelerator object.
    args (`TrainingArguments`): The training arguments to propagate to DeepSpeed config.
    auto_find_batch_size (`bool`, *optional*, defaults to `False`):
        Whether batch size was auto-discovered by trying increasingly smaller sizes.
N)r(  r)  r-   r*  rF   r‹   rg   )r'  re   rf   Ú	ds_plugins   &&& r   Úpropagate_args_to_deepspeedrA  ¢  sV   € ð ×!Ñ!×2Ñ2€Iä5°i×6LÑ6L×6SÑ6SÓT€IÔØ!*×!7Ñ!7×!>Ñ!>€IÔØ×Ñ×1Ñ1°$ÖMr    c                ó  aa€ RV9  d   RV9   d   VR,          VR&   V! R	/ VB pVP                   pVP                  R8X  d)   VP                  ^8”  d   ^ RIHp VP                  4       pM;V P                  e#   V P                  R,          P                  4       pM\        R4      hVP                  p	\        P                  P                  P                  P                  WhR7      oVR,          R
8g  P                  R4      P                  4       p
\        P                  P                  P                  P                  W¨R7      o\        VV3R l\!        V	4       4       4      p\        S4      pV\#        V^4      ,          pV'       d   We3# T# )aA  
Computes the loss under sequence parallelism with `sp_backend="deepspeed"` and `sp_size > 1`.

Performs weighted loss aggregation across SP ranks, accounting for varying numbers of valid tokens per rank
(e.g., when some ranks receive only padding or prompt tokens that are masked with -100).

Args:
    accelerator (`Accelerator`): The accelerator instance with `torch_device_mesh` support.
    model (`torch.nn.Module`): The model to compute the loss for.
    inputs (`dict[str, torch.Tensor | Any]`): The input data for the model. Must include `"shift_labels"` key.
    return_outputs (`bool`): Whether to return the model outputs along with the loss.
    pc (`accelerate.parallelism_config.ParallelismConfig`): The parallelism configuration.

Returns:
    The loss, or a tuple of `(loss, outputs)` if `return_outputs` is `True`.
ÚlabelsÚshift_labelsr	   )ÚgroupsÚspz™Sequence parallelism is enabled but no SP process group is available. Ensure torch_device_mesh is initialized or sp_backend='deepspeed' with sp_size > 1.)Úgroupc              3   ór   <"  € T F,  pSV,          ^ 8”  g   K  SV,          SV,          ,          x € K.  	  R# 5i)r4  Nr!   )Ú.0ÚrankÚgood_tokens_per_rankÚlosses_per_ranks   & €€r   Ú	<genexpr>Ú,deepspeed_sp_compute_loss.<locals>.<genexpr>â  s:   øé € ð á(ˆDØ Õ%¨Ñ)ô 	;ˆ˜ÕÐ 4°TÕ :×:Ò:Û(ùs   ƒ7˜7r!   iœÿÿÿrð   )ÚlossÚ
sp_backendÚsp_sizer&  rE  Ú_get_sequence_parallel_groupÚtorch_device_meshÚ	get_groupr4   ra   rô   r   Ú
functionalÚ
all_gatherÚviewÚsumÚrangert   )r'  ry   ÚinputsÚreturn_outputsÚpcÚoutputsrO  rE  Úsp_groupÚsp_world_sizeÚgood_tokensÚ
total_lossÚtotal_good_tokensrK  rL  s   &&&&&        @@r   Údeepspeed_sp_compute_lossrc  ³  s\  ù€ ð, �vÔ .°FÔ":à! .Õ1ˆˆxÑÙ‰o�f‰o€GØ�<‰<€Dð 
‡}�}˜Ô#¨¯
©
°Q¬Ý*à×6Ñ6Ó8‰Ø	×	&Ñ	&Ò	2Ø×0Ñ0°Õ6×@Ñ@ÓB‰äðbó
ð 	
ð —J‘J€Mä×'Ñ'×*Ñ*×5Ñ5×@Ñ@ÀÐ@ÓV€Oà˜.Õ)¨TÑ1×7Ñ7¸Ó;×?Ñ?ÓA€KÜ ×,Ñ,×/Ñ/×:Ñ:×EÑEÀkÐEÓbÐäõ ä˜-Ô(óó €Jô
 Ð0Ó1ÐØœÐ-¨qÓ1Õ1€Dç,ˆDˆ?Ð6°$Ð6r    r/   r€   )T)'r&   r  Úimportlib.metadatar
   Úimportlib.utilr‚   Ú	functoolsr   Údependency_versions_checkr   Úutilsr   r   r   ra   r   Ú
get_loggerr"   r  r   Úaccelerate.utils.deepspeedr   ÚDeepSpeedConfigÚbuiltinsr   r-   r„   r   r‡   r‰   r‹   r¦   rã   r
  r   r2  r>  rA  rc  r!   r    r   Ú<module>rm     sÓ   ðñó Û Û Û Ý #å 9ß HÑ Hñ ×ÒÛÝð 
×	Ò	˜HÓ	%€ò
ñ ×ÒÑ!7×!9Ò!9ÞOõ 3ô.˜ô .ô2pÐ0ô pðh !%Ð òIò)òòò ;òFlô^W$òt3#ôl@#ôFPô0Nô"77r    