+
    QV-jÄ ã                   ó¢
  a € R’ Et0 t ^ RIt^ RIt^ RIt^ RIt^ RIt^ RIt^ RIt^ RIt^ RI	t	^ RI
Ht ^ RIHt ^ RIHtHt ^ RIHt ^ RIHtHt ^ RIHtHt ^ RIHt ^ R	IHt ^ R
IHtHtHtHtH t  ^ RI!H"t" ^ RI#t#^ RI$H%t%H&t& ^ RI'H(t( ^ RI)H*t* ^ RI+H,t- ^ RI+H.t/ ^ RI#H0t0H1t1 ^ RI2H3t3 ^ RI4H5t5 ^RI6H7t8 ^RI9H:t: ^RI;H<t< ^RI=H>t>H?t?H@t@HAtA ^RIBHCtC ^RIDHEtE ^RIFHGtGHHtH ^RIIHJtJHKtKHLtLHMtMHNtN ^RIOHPtPHQtQHRtRHStSHTtTHUtUHVtV ^RIWHXtX ^RIYHZtZ ^RI[H\t\ ^R I]H^t^ ^R!I_H`t` ^R"IaHbtb ^R#IcHdtdHete ^R$IfHgtg ^R%IhHiti ^R&IjHktk ^R'IlHmtm ^R(InHotoHptpHqtqHrtrHstsHtttHutu ^R)IvHwtw ^R*IxHytyHztzH{t{H|t| ^R+I}H~t~ ^R,IH€t€H�t� ^R-I‚Hƒtƒ ^R.I„H…t… ^R/I†H‡t‡ ^R0IˆH‰t‰ ^R1IŠH‹t‹ ^R2IŒH�t�HŽtŽH�t�H�t�H‘t‘H’t’H“t“H”t”H•t•H–t–H—t—H˜t˜H™t™HštšH›t›HœtœH�t�HžtžHŸtŸH t H¡t¡ ^R3I¢H£t£H¤t¤H¥t¥ ^R4I¦H§t§H¨t¨H©t©Hªtª ^R5I«H¬t¬H­t­H®t®H¯t¯H°t° ^R6I±H²t²H³t³ ^R7I´HµtµH¶t¶ ^R8I·H¸t¸ ]š! 4       '       d   ^ R9I¹Hºtº ^ R:I»H¼t¼ ]'       d   ^R;I½H¾t¾ ]#P„                  EP                  4       tÀ]®! 4       '       d8   ^ RIÁHÂu H#tÃ ^ R<IÄHÅtÆ ](EPŽ                  ! ]Æ4      ](EPŽ                  ! R=4      8¬  tÈMR>tÈ]¡EP’                  ! ]Ê4      tË]EP˜                  EP›                  R?R@4      EP�                  4       tÏ]EP˜                  EP›                  RAR@4      EP�                  4       tÐ]! RBRCRD7      tÑR>sÒR>sÓ]! RERF7       ! RG RH4      4       tÔRI RJ ltÕRK RL ltÖRM t×]RN 4       tØ]RO 4       tÙ]R“RP RQ ll4       tÚRR tÛRS tÜRT]#EPº                  RU]#EP¼                  RV]#EP¾                  RW]#EPÀ                  RX]#EPÂ                  RY]#EPÄ                  RZ]#EPÆ                  R[]#EPÈ                  R\]#EPÊ                  R]]#EPÌ                  R^]#EPÎ                  R_]#EPÐ                  R`]#EPÒ                  Ra]#EPÔ                  Rb]#EPÖ                  /tìRc Rd ltíR”Re Rf lltîRg Rh ltïRi Rj ltðRk Rl ltñRm Rn ltòRo Rp ltóRq Rr ltôR“Rs Rt lltõR•Ru Rv lltöR“Rw Rx llt÷ ! Ry Rz4      tø ! R{ R|4      tù ! R} RC]1EPô                  ]ù]ø]•]J4      tû]˜! ]ûEPø                  4      ]ûnü        ]ûEPø                  EPú                  e<   ]ûEPø                  EPú                  EPý                  R~RR€R�7      ]ûEPø                  ný        ] R–R‚ Rƒ ll4       tÿ] R–R„ R… ll4       tÿR–R† R‡ lltÿRˆ R‰ lEt R“RŠ R‹ llEtRŒ R� lEt ! RŽ R�]£4      EtE]! 4       Et] ^ k  ! R� R‘]û4      EtR# )—é    N)Úabstractmethod)Údefaultdict)ÚCallableÚIterator)Úcontextmanager)Ú	dataclassÚfield)ÚpartialÚwraps)Úcycle)ÚThread)ÚTYPE_CHECKINGÚAnyÚTypeVarÚget_type_hintsÚoverload)Ú
is_zipfile)Úis_offline_modeÚ"split_torch_state_dict_into_shards)Úversion)Ú	safe_open)Úload)Ú	save_file)ÚTensorÚnn)Úconstraints)Ú
checkpoint)Úinitialization©ÚPreTrainedConfig)Úget_model_conversion_mapping)ÚWeightConverterÚWeightRenamingÚ$convert_and_load_state_dict_in_modelÚrevert_weight_conversion)ÚDistributedConfig)Úcustom_object_save)ÚCompileConfigÚGenerationConfig)ÚPeftAdapterMixinÚdeepspeed_configÚhub_kernelsÚis_deepspeed_zero3_enabledÚis_fsdp_enabled)Ú_get_device_mapÚaccelerate_disk_offloadÚaccelerate_dispatchÚcheck_and_set_device_mapÚexpand_device_mapÚ
get_deviceÚload_offloaded_parameter)Ú!_load_state_dict_into_zero3_model)Úeager_paged_attention_forward)ÚALL_FP8_EXPERTS_FUNCTIONS)Úflash_attention_forward)Úpaged_attention_forward)Úflex_attention_forward)Úallow_all_hub_kernelsÚ	is_kernel)ÚALL_EXPERTS_FUNCTIONS)Úmaybe_load_adapters)Úsdpa_attention_forward)Úsdpa_attention_paged_forward)ÚALL_PARALLEL_STYLESÚ_get_parameter_tp_planÚdistribute_modelÚgather_state_dict_for_saveÚinitialize_tensor_parallelismÚshard_and_distribute_moduleÚverify_tp_plan)ÚLOSS_MAPPING)Ú$FLASH_ATTENTION_COMPATIBILITY_MATRIXÚFLASH_ATTN_KERNEL_FALLBACKÚlazy_import_flash_attentionÚ!lazy_import_paged_flash_attention)ÚROPE_INIT_FUNCTIONS)Úapply_patchesÚpatch_output_recorders)Úid_tensor_storage)ÚHfQuantizer)Úget_hf_quantizer)Úget_module_from_name)Úauto_conversion)ÚADAPTER_SAFE_WEIGHTS_NAMEÚDUMMY_INPUTSÚSAFE_WEIGHTS_INDEX_NAMEÚSAFE_WEIGHTS_NAMEÚWEIGHTS_INDEX_NAMEÚWEIGHTS_NAMEÚContextManagersÚKernelConfigÚPushToHubMixinÚcached_fileÚcheck_torch_load_is_safeÚ	copy_funcÚhas_fileÚis_accelerate_availableÚis_bitsandbytes_availableÚis_env_variable_trueÚis_kernels_availableÚis_torch_flex_attn_availableÚis_torch_npu_availableÚis_torch_xpu_availableÚlogging)ÚGeneralInterfaceÚis_flash_attention_requestedÚsplit_attention_implementation)ÚDownloadKwargsÚcreate_and_tag_model_cardÚget_checkpoint_shard_filesÚhf_api)Úis_flash_attn_greater_or_equalÚ#is_huggingface_hub_greater_or_equalÚis_sagemaker_mp_enabledÚis_torch_cuda_availableÚ
is_tracing)ÚLoadStateDictInfoÚlog_state_dict_report)Ú_CAN_RECORD_REGISTRYÚOutputRecorder)ÚQuantizationMethod)Úadd_hook_to_module)Úextract_model_from_parallel)ÚDeviceMeshLike)Ú__version__z1.10FÚXLA_USE_BF16Ú0ÚXLA_DOWNCAST_BF16ÚSpecificPreTrainedModelTypeÚPreTrainedModel)ÚboundT)Úfrozenc                   ó¤   a € ] tR t^¦t o RtRt]! ]R7      tRt	Rt
RtRtRtRtRt]! ]R7      tRtRtRtRtRt]V 3R lR l4       tV 3R ltR	tV tR# )
ÚLoadStateDictConfigzY
Config for loading weights. This allows bundling arguments that are just
passed around.
N)Údefault_factoryFTc                ó    <€ V ^8„  d   QhRS[ /# ©é   Úreturn©Úbool)ÚformatÚ__classdict__s   "€Úl/Volumes/fast/ai/experiments/ui-tars-smoke/.venv/lib/python3.14/site-packages/transformers/modeling_utils.pyÚ__annotate__Ú LoadStateDictConfig.__annotate__¾   s   ø€ ÷ -ñ -™dñ -ó    c                ó   € V P                   R J# ©N)Úhf_quantizer©Úselfs   &r’   Úis_quantizedÚ LoadStateDictConfig.is_quantized½   s   € à× Ñ ¨Ð,Ð,r•   c                óŒ  <€ V ^8„  d   Qh/ S[ R,          ;R&   S[R,          ;R&   S[R,          ;R&   S[;R&   S[R,          ;R&   S[R,          ;R&   S[ R,          ;R&   S[;R	&   S[P
                  R,          ;R
&   S[;R&   S[R,          ;R&   R;R&   S[;R&   S[S[S[	,          ,          R,          ;R&   S[R,          ;R&   # )rŒ   NÚpretrained_model_name_or_pathÚdownload_kwargsÚuse_safetensorsÚignore_mismatched_sizesÚsharded_metadataÚ
device_mapÚdisk_offload_folderÚoffload_buffersÚdtypeÚ
dtype_planr˜   úDeviceMeshLike | NoneÚdevice_meshÚweights_onlyÚweight_mappingÚdisable_mmap)
Ústrrn   r�   ÚdictÚtorchr¦   rR   Úlistr"   r#   )r�   r‘   s   "€r’   r“   r”   ¦   s  ø‡ ‚ ñ $'¨¥:Ñ4ñ ñ $ dÕ*ÑRñ ñ ˜D•[Ñ'ñ ñ "Ñ)ñ ñ ˜T•kÑ(ñ ñ �t•Ñ"ñ ñ ˜t�Ñ*ñ ñ Ñ!ñ ñ �;‰;˜ÕÑ$ñ ñ  Ñ2ñ! ñ"  Õ$Ñ+ñ# ð$ )Ñ/ñ% ñ& Ññ' ñ( ™©>Õ9Õ:¸TÕAÑHñ) ñ* ˜•+Ñ$ò+ r•   © )Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__rž   r	   rn   rŸ   r    r¡   r¢   r£   r¤   r¥   r¦   r®   r§   r˜   r©   rª   r«   r¬   Úpropertyr›   Ú__annotate_func__Ú__static_attributes__Ú__classdictcell__©r‘   s   @r’   rˆ   rˆ   ¦   s~   ø‡ € ñð
 15Ð!Ù-2À>Ô-R€OØ#'€OØ$)ÐØ$(ÐØ"€JØ&*ÐØ!€OØ $€EÙ¨TÔ2€JØ'+€LØ+/€KØ€LØDH€NØ $€Là÷-ó ð-÷1 ƒ r•   rˆ   c                ó$   € V ^8„  d   QhR\         /# r‹   rŽ   )r�   s   "r’   r“   r“   Â   s   € ÷ ñ ¬4ñ r•   c                  óž   € \         ;'       dA    \        \        P                  R 4      ;'       d    \        P                  P	                  4       # )Úis_initialized)Ú_torch_distributed_availableÚhasattrr¯   Údistributedr¾   r±   r•   r’   Ú!_is_torch_distributed_initializedrÂ   Â   sA   € ä$÷ 	/ð 	/Ü”E×%Ñ%Ð'7Ó8÷	/ð 	/ä×Ñ×,Ñ,Ó.ðr•   c                ó$   € V ^8„  d   QhR\         /# r‹   ©Úint)r�   s   "r’   r“   r“   Ê   s   € ÷ .ñ .¬3ñ .r•   c                  ó¢   € \        4       '       d!   \        \        P                  R 4      '       g   ^# \        P                  P	                  4       # )Úget_world_size)rÂ   rÀ   r¯   rÁ   rÇ   r±   r•   r’   Ú!_get_torch_distributed_world_sizerÈ   Ê   s6   € Ü,×.Ò.´g¼e×>OÑ>OÐQa×6bÒ6bÙÜ×Ñ×+Ñ+Ó-Ð-r•   c                  ó~   € \        4       ;'       d-    \        \        P                  P	                  R R4      4      ^ 8H  # )Ú
LOCAL_RANKz-1)rÂ   rÅ   ÚosÚenvironÚgetr±   r•   r’   Úis_local_dist_rank_0rÎ   Ð   s.   € Ü,Ó.×_Ð_´3´r·z±z·~±~ÀlÐTXÓ7YÓ3ZÐ^_Ñ3_Ð_r•   c               #   ó.   "  € R s  Rx € Rs R#   Rs i ; i5i©TNF)Ú_is_quantizedr±   r•   r’   Úset_quantized_staterÒ   Ô   s   é € ð €MðÛàŠø˜‰üó   ‚† ŠŽ’c               #   ó.   "  € R s  Rx € Rs R#   Rs i ; i5irÐ   )Ú_is_ds_init_calledr±   r•   r’   Úset_zero3_staterÖ   á   s!   é € ð Ðð#Ûà"Òø˜UÑürÓ   c                óR   € V ^8„  d   QhR\         P                  R\        R,          /# )rŒ   r¦   Úmodel_class_nameN)r¯   r¦   r­   )r�   s   "r’   r“   r“   ì   s"   € ÷ 0ñ 0œUŸ[™[ð 0¼CÀ$½Jñ 0r•   c              #  ó0  "  € V P                   '       g   Ve
   V RV  R2pMRV  R2p\        V4      h\        P                  ! 4       p \        P                  ! V 4       Rx € \        P                  ! V4       R#   \        P                  ! T4       i ; i5i)zÔ
Locally change the torch default dtype to `dtype`, and restore the old one upon exiting the context.
If `model_class_name` is provided, it's used to provide a more helpful error message if `dtype` is not valid.
Nz% cannot be instantiated under `dtype=z$` as it's not a floating-point dtypezCannot set `z7` as torch's default as it's not a floating-point dtype)Úis_floating_pointÚ
ValueErrorr¯   Úget_default_dtypeÚset_default_dtype)r¦   rØ   Úerror_messageÚoriginal_dtypes   &&  r’   Úlocal_torch_dtyperà   ë   s�   é € ð ×"×"Ð"ØÒ'à#Ð$Ð$IÈ%ÈÐPtÐuñ ð +¨5¨'Ð1hÐiˆMÜ˜Ó'Ð'ä×,Ò,Ó.€Nð0Ü×Ò Ô&Ûä×Ò Ö/øŒ×Ò Õ/üs   ‚ABÁ	A; Á#BÁ;BÂBc                 óº   € \         P                  ! . 4      P                  p \         P                  ! 4       pW8X  d    V\         P                  ! R4      8w  d   V# R# V # )zà
Test if a device context manager is currently in use, or if it is not the case, check if the default device
is not "cpu". This is used to infer the correct device to load the model on, in case `device_map` is not provided.
ÚcpuN)r¯   ÚtensorÚdeviceÚget_default_device)Údevice_in_contextÚdefault_devices     r’   Ú*get_torch_context_manager_or_global_devicerè     sM   € ô
 Ÿš RÓ(×/Ñ/ÐÜ×-Ò-Ó/€NàÔ*ØœUŸ\š\¨%Ó0Ô0Ø!Ð!ÙØÐr•   c                ó€  € V P                  4        F_  pVP                  4       '       g   K  R\        VP                  4      9  g   K7  R\        VP                  4      9  g   KS  VP                  u # 	  \	        V 4      ^ 8X  d   \
        P                  # \        \        V P                  4       4      4      P                  # )zl
Returns the first found floating dtype in `state_dict` if there is one, otherwise returns the first dtype.
Úfloat8_Úfloat4_)	ÚvaluesrÚ   r­   r¦   Úlenr¯   Úfloat32ÚnextÚiter)Ú
state_dictÚts   & r’   Úget_state_dict_dtyperó     s‰   € ð ×ÑÖ ˆà×Ñ× Ô  Y´c¸!¿'¹'³lÖ%BÀyÔX[Ð\]×\cÑ\cÓXdÖGdØ—7‘7ŠNñ !ô ˆ:ƒ˜!ÔÜ�}‰}ÐÜ”�Z×&Ñ&Ó(Ó)Ó*×0Ñ0Ð0r•   ÚBOOLÚU8ÚI8ÚI16ÚU16ÚF16ÚBF16ÚI32ÚU32ÚF32ÚF64ÚI64ÚU64ÚF8_E4M3ÚF8_E5M2c                ó(   € V ^8„  d   QhRRR\         /# )rŒ   Úpathzstr | os.PathLiker�   rŽ   )r�   s   "r’   r“   r“   4  s   € ÷ ñ Ð-ð ´$ñ r•   c                ó  € \         P                  P                  R4      '       g   R#  \        P                  P                  \        P                  ! V 4      4      p\        RRR7      ;_uu_ 4       p\        R R V 4        4       R R	R
7      pRRR4       X F?  w  rEW8X  g0   VP                  VP                  R4      R,           4      '       g   K:  VR8H  u # 	  R#   + '       g   i     LW; i  \        \        3 d     R# i ; i)a  True if `path` lives on an hf-mount FUSE filesystem (device string 'hf-mount').

hf-mount's mmap + readahead interaction deadlocks under parallel page-faults,
so callers should load the file into memory instead. Linux-only; returns False
on other platforms.
ÚlinuxFz/proc/mountsúutf-8©Úencodingc              3   ój   "  € T F)  p\        V4      ^8¼  g   K  V^ ,          V^,          3x € K+  	  R# 5i)rŒ   N©rí   )Ú.0Úps   & r’   Ú	<genexpr>Ú"_is_on_hf_mount.<locals>.<genexpr>A  s*   é € ÐNÑ'> !Ä#ÀaÃ&ÈAÁ+”�!�A•$˜˜!�•Ó'>ùs   ‚3™3c              3   ó@   "  € T F  qP                  4       x € K  	  R # 5ir—   )Úsplit)r  Úls   & r’   r  r  A  s   é € Ð'>¹2°a¯©¯	¨	»2ùs   ‚c                 ó&   € \        V ^,          4      # )é   r  )Úes   &r’   Ú<lambda>Ú!_is_on_hf_mount.<locals>.<lambda>B  s   € œc ! A¥$œir•   T)ÚkeyÚreverseNÚ/zhf-mount)ÚsysÚplatformÚ
startswithrË   r  ÚrealpathÚfspathÚopenÚsortedÚrstripÚOSErrorrÛ   )r  ÚrealÚfhÚentriesÚdevÚmps   &     r’   Ú_is_on_hf_mountr)  4  sÑ   € ô �<‰<×"Ñ" 7×+Ò+ÙðÜ�w‰w×Ñ¤§	¢	¨$£Ó0ˆÜ�.¨7×3Õ3°rÜÙNÑ'>¹2Ô'>ÓNÙ'ØôˆG÷ 4ó ‰GˆCØŒz˜TŸ_™_¨R¯Y©Y°s«^¸cÕ-A×BÔBØ˜jÑ(Ò(ñ ñ
 ÷ 4×3ûô ”ZÐ ô ØÙðús6   ©AC1 Á1CÂ?C1 ÃC1 ÃC1 ÃC.	Ã)C1 Ã1DÄDc                óì   € V ^8„  d   QhR\         \        P                  ,          R\         \        P                  ,          R\
        R\
        R,          R\        \         \        P                  3,          /# )rŒ   Úcheckpoint_fileÚmap_locationrª   r¬   Nr�   )r­   rË   ÚPathLiker¯   rä   r�   r®   r   )r�   s   "r’   r“   r“   M  se   € ÷ /kñ /kÜœ2Ÿ;™;Õ&ð/käœŸ™Õ$ð/kô ð/kô ˜•+ð	/kô
 
Œ#Œu�|‰|Ð
Õñ/kr•   c           	     ó  € \         P                  ! V 4      pVf   \        V4      pVP                  R4      '       EdX   V'       dy   VR8w  dr   \	        VR4      ;_uu_ 4       p\        VP                  4       4      pRRR4       VR8w  d3   XP                  4        UUu/ uF  w  rxWxP                  V4      bK  	  pppX# \        VRR7      ;_uu_ 4       p	/ pV	P                  4        FŸ  pVR8X  dt   V	P                  V4      p
V
P                  4       pV\        9   d   \        V,          pM\        RV 24      h\        P                   ! V
P#                  4       VRR	7      Wg&   K}  V	P%                  V4      P                  V4      Wg&   K¡  	  VuuRRR4       # V'       d   \'        4        / pVR8w  d   \)        V4      '       d   R
R/p\        P*                  ! V3RVRV/VB #   + '       g   i     ELn; iu uppi   + '       g   i     Lu; i)aC  
Reads a `safetensor` or a `.bin` checkpoint file. We load the checkpoint on "cpu" by default.

When `disable_mmap` is True, safetensors files are read fully into memory instead of
being memory-mapped. When `disable_mmap` is None (default), it is auto-detected to True
on hf-mount FUSE filesystems (see `_is_on_hf_mount`).
Nú.safetensorsÚmetaÚrbrâ   Úpt)Ú	frameworkz)Cannot load safetensors of unknown dtype )Úsizer¦   rä   ÚmmapTr,  rª   )rË   r  r)  Úendswithr   Ú_safe_load_bytesÚreadÚitemsÚtor   ÚkeysÚ	get_sliceÚ	get_dtypeÚstr_to_torch_dtyperÛ   r¯   ÚemptyÚ	get_shapeÚ
get_tensorr`   r   r   )r+  r,  rª   r¬   Úcheckpoint_pathÚ_fhrñ   ÚkÚvÚfÚ_sliceÚk_dtyper¦   Ú
extra_argss   &&&&          r’   Úload_state_dictrJ  M  s²  € ô —i’i Ó0€OØÒÜ& Ó7ˆà×Ñ ×/Ó/ß˜L¨FÔ2Ü�o t×,Ô,°Ü-¨c¯h©h«jÓ9�
÷ -à˜uÔ$Ø@J×@PÑ@PÔ@RÔSÑ@R¹¸˜a§¡ lÓ!3Ò3Ñ@R�
ÑSØÐÜ�°$×7Õ7¸1ØˆJØ—V‘V–X�Ø 6Ô)ØŸ[™[¨›^�FØ$×.Ñ.Ó0�GØÔ"4Ô4Ü 2°7Õ ;™ä(Ð+TÐU\ÐT]Ð)^Ó_Ð_Ü$)§K¢K°V×5EÑ5EÓ5GÈuÐ]cÔ$d�J“Mà$%§L¡L°£O×$6Ñ$6°|Ó$D�J“Mñ ð ÷ 8Ò7÷  Ü Ô"Ø€Jà�vÔ¤*¨_×"=Ò"=Ø˜d�^ˆ
ä�:Š:�oÑj°LÐjÈ|ÐjÐ_iÑjÐj÷9 -×,Ð,üó Tç7×7ús   Á!G ÂG4ÃB7G:Ç G1	Ç:H
	c                óD   € V ^8„  d   QhR\         P                  R\        /# )rŒ   rã   r�   )r¯   r   rÅ   )r�   s   "r’   r“   r“     s   € ÷ ñ ”U—\‘\ð ¤cñ r•   c                 óÌ   € V P                  4       '       d>   V P                  R4      R,          P                  4       V P                  4       ,           pV# V P                  4       pV# )r  éÿÿÿÿ)ÚnelementÚviewÚdata_ptrÚelement_size)rã   Ústops   & r’   Ú_end_ptrrS    sR   € à‡�×ÒØ�{‰{˜2‹˜rÕ"×+Ñ+Ó-°×0CÑ0CÓ0EÕEˆð €Kð �‰Ó ˆØ€Kr•   c                óZ   € V ^8„  d   QhR\         P                  R\        \        ,          /# )rŒ   Úmoduler�   )r   ÚModuler°   r­   )r�   s   "r’   r“   r“   ˆ  s"   € ÷ ñ ¤"§)¡)ð ´´Sµ	ñ r•   c           	      óî   € . pV P                  4        FY  w  r#\        VR / 4      ;'       g    / pTP                  VP                  4        Uu. uF  qR'       d   V RV 2MTNK  	  up4       K[  	  V# u upi )Ú_tied_weights_keysÚ.)Únamed_modulesÚgetattrÚextendr;  )rU  Útied_weight_keysÚnameÚ	submoduleÚtiedrD  s   &     r’   Ú_get_tied_weight_keysra  ˆ  sv   € Ø"$ÐØ!×/Ñ/Ö1‰ˆÜ�yÐ"6¸Ó;×AÐA¸rˆØ×ÑÀtÇyÁyÄ{Ó SÁ{À!¶$ D 6¨¨1¨#¡¸AÒ!=Á{Ñ SÖTñ 2ð Ðùò !Ts   ÁA2
c          	      ó  € V ^8„  d   QhR\         \        \        ,          ,          R\        \        \        P
                  3,          R\        \         \        \        ,          ,          \         \        ,          3,          /# ©rŒ   Útensorsrñ   r�   ©r°   Úsetr­   r®   r¯   r   Útuple)r�   s   "r’   r“   r“   �  sV   € ÷ ,ñ ,œD¤¤S¥�Nð ,¼¼SÄ%Ç,Á,Ð=NÕ8Oð ,ÔTYÔZ^Ô_bÔcfÕ_gÕZhÔjnÔorÕjsÐZsÕTtñ ,r•   c                 óf  € . pV  FØ  p\        V4      ^8  d   VP                  V4       K&  . pV F6  pW,          pVP                  VP                  4       \        V4      V34       K8  	  VP	                  4        V^ ,          w  rxp	VP                  V	04       VR,           F9  w  r«pW¨8¼  d   VP                  V04       MVR,          P                  V4       TpK;  	  KÚ  	  . p. pV FE  p \        V 4      ^8X  d"   VP                  V P                  4       4       K4  VP                  V 4       KG  	  WÜ3# )rŒ   :r  NNrM  )rí   ÚappendrP  rS  ÚsortÚaddÚpop)rd  rñ   Úfiltered_tensorsÚsharedÚareasr^  rã   Ú_Ú	last_stopÚ	last_nameÚstartrR  Údisjoint_tensorsÚshared_tensorss   &&            r’   Ú_find_disjointrv  �  s   € ØÐÛˆÜˆv‹;˜Œ?Ø×#Ñ# FÔ+ÙàˆÛˆDØÕ%ˆFØ�L‰L˜&Ÿ/™/Ó+¬X°fÓ-=¸tÐDÖEñ ð 	�
‰
Œà"'¨¥(Ñˆ�iØ×Ñ  Ô,Ø!& r§ ÑˆE˜ØÔ!Ø ×'Ñ'¨¨Õ/à  Õ$×(Ñ(¨Ô.ØŠIó "+ñ ð& ÐØ€NÛ#ˆÜˆw‹<˜1ÔØ×#Ñ# G§K¡K£MÖ2à×!Ñ! 'Ö*ñ	 $ð
 Ð+Ð+r•   c          
      ó  € V ^8„  d   QhR\         \        \        ,          ,          R\        \        \        P
                  3,          R\        \         \        \        ,          ,          \         \        \        ,          ,          3,          /# rc  re  )r�   s   "r’   r“   r“   ¯  sT   € ÷ %ñ %Ü”#”c•(�^ð%Ü)-¬c´5·<±<Ð.?Õ)@ð%ä
Œ4””C•�>œ4¤¤C¥�>Ð)Õ*ñ%r•   c                 ó~  € . p. pV  F±  p\        V4      ^8  d   K  \        P                  ! \        4      pV FH  pW,          pVP                  VP                  4       \        V4      3pWX,          P                  V4       KJ  	  \        V4      ^8X  d   VP                  V4       K   VP                  V4       K³  	  W#3# )rŒ   )	rí   Úcollectionsr   rf  rä   rP  rS  rk  ri  )	rd  rñ   ru  Ú	identicalrn  ro  r^  rã   Úareas	   &&       r’   Ú_find_identicalr|  ¯  s¦   € ð €NØ "€IÛˆÜˆv‹;˜Œ?Ùä×'Ò'¬Ó,ˆÛˆDØÕ%ˆFØ—M‘M 6§?¡?Ó#4´h¸vÓ6FÐGˆDØ�K�O‰O˜DÖ!ñ ô ˆu‹:˜Œ?Ø×Ñ˜VÖ$à×!Ñ! &Ö)ñ ð Ð$Ð$r•   c                ó    € V ^8„  d   QhR\         \        \        P                  3,          RRR\         \        \        P                  3,          /# )rŒ   rñ   Úmodelr„   r�   ©r®   r­   r¯   r   )r�   s   "r’   r“   r“   Ä  sE   € ÷ Mñ MÜ”Sœ%Ÿ,™,Ð&Õ'ðMØ0AðMä	Œ#Œu�|‰|Ð
ÕñMr•   c                óž  a€ \         P                  ! \        4      pV P                  4        F¹  w  op\	        V\
        P                  4      '       g$   V\        V4      ,          P                  S4       KI  VP                  P                  R8X  d5   VP                  S4      pV\        V4      ,          P                  S4       K˜  V\        V4      ,          P                  S4       K»  	  VP                  4        UUu/ uF  w  rE\        V4      ^8”  g   K  WEbK  	  ppp\        \        V4      4      p. p\        4       p	Ve¥   VP!                  4        F�  p^ p
\#        V4       F|  o\$        ;QJ d    V3R lV 4       F  '       g   K   RM	  RM! V3R lV 4       4      pV'       g   KG  SV 9   g   KP  V
^,          p
V
\        V4      8  g   Kk  V	P'                  S4       K~  	  K’  	  \)        VP!                  4       V 4      w  rÍV F  oV S,          P+                  4       V S&   K  	  \-        WÀ4      w  rÎV FT  pVP/                  V	4      pV F  oV S K  	  VP1                  V	4      p\        V4      ^8”  g   KC  VP                  V4       KV  	  V'       d   VP3                  V4       \        V4      ^ 8”  d   \5        RV RV R24      hV # u uppi )aH  
Remove all tied weights from the given `state_dict`, making sure to keep only the main weight that `model`
will expect when reloading (even if we now tie weights symmetrically, it's better to keep the intended one).
This is because `safetensors` does not allow tensor aliasing - so we're going to remove aliases before saving.
r0  c              3   óR   <"  € T F  p\         P                  ! VS4      x € K  	  R # 5ir—   ©ÚreÚsearch)r  Úpatr^  s   & €r’   r  Ú6remove_tied_weights_from_state_dict.<locals>.<genexpr>ì  s!   øé € Ð%fÑFe¸s¤b§i¢i°°T×&:Ð&:ÓFeùs   ƒ$'TFz8The weights trying to be saved contained shared tensors z\ which are not properly defined. We found all the potential target tied weights keys to be: zo.
This can also just mean that the module's tied weight keys are wrong vs the actual tied weights in the model.)ry  r   r°   r9  Ú
isinstancer¯   r   Úidri  rä   ÚtypeÚget_parameterrQ   rí   rf  ra  rì   r!  Úanyrk  rv  Úcloner|  ÚintersectionÚ
differencer\  ÚRuntimeError)rñ   r~  Úptrsrã   ÚptrÚnamesÚshared_ptrsÚall_potential_tied_weights_keysÚerror_namesÚto_delete_namesÚfoundÚmatches_patternÚshared_namesÚdisjoint_namesÚidentical_namesÚinamesÚknownÚunknownr^  s   &&                @r’   Ú#remove_tied_weights_from_state_dictrŸ  Ä  sv  ø€ ô ×"Ò"¤4Ó(€DØ"×(Ñ(Ö*‰ˆˆfÜ˜&¤%§,¡,×/Ò/ð ”�F“Õ×#Ñ# DÖ)à�]‰]×Ñ 6Ô)ð ×(Ñ(¨Ó.ˆFØ”�F“Õ×#Ñ# DÖ)ð Ô" 6Ó*Õ+×2Ñ2°4Ö8ñ +ð" 15·
±
´ÔO±¡* #ÄÀEÃ
ÈQÁ”:�3’:±€KÑOô '*Ô*?ÀÓ*FÓ&GÐ#Ø€KÜ“e€Oð 'Ò2Ø ×'Ñ'Ö)ˆEØˆEÜ˜už�ß"%£#Ô%fÑFeÓ%f§#§#¢#Ô%fÑFeÓ%fÓ"f�ß"‘? t¨zÖ'9Ø˜Q•J�EØœs 5›zÖ)Ø'×+Ñ+¨DÖ1ó &ñ *ô $2°+×2DÑ2DÓ2FÈ
Ó#SÑ €Ló ˆØ% dÕ+×1Ñ1Ó3ˆ
�4Óñ ô %4°LÓ$MÑ!€Lã!ˆØ×#Ñ# OÓ4ˆÛˆDØ˜4Ò ñ à×#Ñ# OÓ4ˆÜˆw‹<˜!ÖØ×Ñ˜wÖ'ñ "÷ Ø×Ñ˜<Ô(ä
ˆ;Ó˜!ÔÜØFÀ{Àmð TJØJiÐIjð k|ð|ó
ð 	
ð Ðùóc Ps   Ã<K	ÄK	c                óH   € V ^8„  d   QhRRR\         R\        P                  /# )rŒ   r~  r„   Ú
param_namerã   )r­   r¯   r   )r�   s   "r’   r“   r“     s)   € ÷ (ñ (Ð&7ð (ÄSð (ÔRW×R^ÑR^ñ (r•   c                óâ   € \        W4      w  r4WCP                  9   dF   \        V\        P                  4      '       g&   \        P                  ! W"P                  4       R7      p\        W4V4       R# )zUCast a single parameter or buffer `param_name` into the `model`, with value `tensor`.)Úrequires_gradN)rT   Ú_parametersr‡  r   Ú	ParameterrÚ   Úsetattr)r~  r¡  rã   ÚparentÚ
param_types   &&&  r’   Ú_load_parameter_into_modelr©    sN   € ä-¨eÓ@Ñ€FØ×'Ñ'Ô'´
¸6Ä2Ç<Á<×0PÒ0PÜ—’˜f×4LÑ4LÓ4NÔOˆô ˆF Ö'r•   c                óJ   € V ^8„  d   QhR\         R\         R,          R\         /# )rŒ   Úweights_nameÚvariantNr�   ©r­   )r�   s   "r’   r“   r“     s%   € ÷ ñ œsð ¬S°4­Zð Ä3ñ r•   c                 óJ   € Ve   V P                  R^4      w  r#V RV RV 2p V # )NrY  )Úrsplit)r«  r¬  r  r^  s   &&  r’   Ú_add_variantr°    s8   € ØÒØ!×(Ñ(¨¨aÓ0‰
ˆØ˜˜q  	¨¨4¨&Ð1ˆØÐr•   c                ó~  € V ^8„  d   QhR\         \        P                  ,          R,          R\         R,          R\         R,          R\        R,          R\        R,          R\        R\         R,          R	\
        R,          R
\        R,          R\        \        \         ,          R,          \        R,          3,          /
# )rŒ   rž   Nr¬  Ú	gguf_filer    Ú
user_agentÚis_remote_codeÚtransformers_explicit_filenamerŸ   Ú
tqdm_classr�   )	r­   rË   r-  r�   r®   rn   r‰  rg  r°   )r�   s   "r’   r“   r“   %  sº   € ÷ I.ñ I.Ü#&¬¯©Õ#4°tÕ#;ðI.ä�4�ZðI.ô �T�zðI.ô ˜D•[ð	I.ô
 �t•ðI.ô ðI.ô %(¨$¥JðI.ô $ dÕ*ðI.ô �t•ðI.ô Œ4”�9�tÕœT D�[Ð(Õ)ñI.r•   c	                ó^  € T;'       g    \        4       pVP                  R4      p	VP                  RR4      p
VP                  R4      pVP                  RR4      pVP                  R4      pVP                  R4      ;'       g    RpVP                  R	R
4      pVP                  R4      pVeD   VP                  R4      '       g-   VP                  R4      '       g   VR8w  d   \        RV 24      hRpV Eeí   VEfè   \	        V 4      p \
        P                  P                  V 4      pV'       Edµ   Ve4   \
        P                  P                  WV4      pVP                  R4      pEM>VRJd‚   \
        P                  P                  \
        P                  P                  W\        \        V4      4      4      '       d1   \
        P                  P                  W\        \        V4      4      pEM·VRJd„   \
        P                  P                  \
        P                  P                  W\        \        V4      4      4      '       d3   \
        P                  P                  W\        \        V4      4      pRpEM.V'       g‚   \
        P                  P                  \
        P                  P                  W\        \        V4      4      4      '       d1   \
        P                  P                  W\        \        V4      4      pEM¥V'       g„   \
        P                  P                  \
        P                  P                  W\        \        V4      4      4      '       d3   \
        P                  P                  W\        \        V4      4      pRpEMV'       d!   \        R\        \        V4       RV  R24      h\        R\        \        V4       R\        \        V4       RV  R24      h\
        P                  P                  \
        P                  P                  Wð4      4      '       d   T pRpEMyVe   TpVP                  R4      pM'VRJd   \        \        V4      pM\        \        V4      pRVRVRVRV	RV/pRV
RVR	VRRRRRVRV/VCp\!        4       '       * ;'       d-    \#        R4      '       * ;'       d    V'       * ;'       d    VR
8H  p \%        V V3/ VB pVf¶   V\        \        V4      8X  d¡   \%        V \        \        V4      3/ VB pVe   RpM~V'       dZ   VR8X  d   V'       d   \'        V 3/ VB w  pppVVR&   Vf1   \        V  R\        \        V4       R\        \        V4       R24      hM\        \        V4      p\%        V V3/ VB pVf7   V\        \        V4      8X  d"   \%        V \        \        V4      3/ VB pVe   RpVeh   V'       d   \        M\        pV\        \        39   dB   \)        V V3/ VB '       g/   V'       d'   \+        \&        V 3R R/VCR!R"7      P-                  4        MmVe:   \)        V \        3/ VB '       d#   \        V  R\        \        V4       R#V R$24      h\        V  R\        \        V4       R\        \        V4       R24      hV'       d   \0        P3                  R(X 24       TpMp\0        P3                  R(X R)X 24       MTV'       dM   \
        P                  P                  V4      '       d   TpM$RV	RV
RVRVRVRVRVR	VRRRRRV/p\%        W3/ VB pRpV'       d   \5        V XV	V
VVVVVVVVR*7      w  ppVV3# V e   X.MRpVV3#   \         d    h \.         d*   p\        R%T  R&T  R'\        \        T4       R24      ThRp?ii ; i)+z´Get all the checkpoint filenames based on `pretrained_model_name_or_path`, and optional metadata if the
checkpoints are sharded.
This function will download the data if necessary.
Ú	cache_dirÚforce_downloadFÚproxiesÚlocal_files_onlyÚtokenÚrevisionÚmainÚ	subfolderÚ Úcommit_hashNr/  z.safetensors.index.jsonzadapter_model.binz¥The transformers file in the config seems to be incorrect: it is neither a safetensors file (*.safetensors) nor a safetensors index file (*.safetensors.index.json): TzError no file named z found in directory rY  z, or z, found in directory r³  Ú _raise_exceptions_for_gated_repoÚ%_raise_exceptions_for_missing_entriesÚ_commit_hashr¶  ÚDISABLE_SAFETENSORS_CONVERSIONz& does not appear to have a file named z or zX and thus cannot be loaded with `safetensors`. Please do not set `use_safetensors=True`.Úignore_errors_during_conversionzThread-auto_conversion)ÚtargetÚargsÚkwargsr^  z) but there is a file without the variant z;. Use `variant=None` to load this model from those weights.zCan't load the model for 'zœ'. If you were trying to load it from 'https://huggingface.co/models', make sure you don't have a local directory with the same name. Otherwise, make sure 'z=' is the correct path to a directory containing a file named zloading weights file z from cache at )
r¸  r¹  rº  r»  r¼  r³  r½  r¿  rÄ  r¶  )rn   rÍ   r6  rÛ   r­   rË   r  ÚisdirÚjoinÚisfiler°  rY   rX   r[   rZ   r#  r   re   r_   rU   rb   r   rs  Ú	ExceptionÚloggerÚinforp   )rž   r¬  r²  r    r³  r´  rµ  rŸ   r¶  r¸  r¹  rº  r»  r¼  r½  r¿  rÁ  Ú
is_shardedÚis_localÚarchive_fileÚfilenameÚhas_file_kwargsÚcached_file_kwargsÚcan_auto_convertÚresolved_archive_fileÚsafe_weights_namer  r¢   Úcheckpoint_filess   &&&&&&&&&                    r’   Ú_get_resolved_checkpoint_filesrÚ  %  s>  € ð &×9Ð9¬Ó)9€OØ×#Ñ# KÓ0€IØ$×(Ñ(Ð)9¸5ÓA€NØ×!Ñ! )Ó,€GØ&×*Ñ*Ð+=¸uÓEÐØ×Ñ Ó(€EØ×"Ñ" :Ó.×8Ð8°&€HØ×#Ñ# K°Ó4€IØ!×%Ñ% mÓ4€KØ%Ò1Ø-×6Ñ6°~×FÒFÐOm×OvÑOvØ%÷P
ò P
ð .Ð1DÔDÜ ð`à5Ð6ð8óð ð €Jà$Ó0°YÓ5FÜ(+Ð,IÓ(JÐ%Ü—7‘7—=‘=Ð!>Ó?ˆçˆ8Ø-Ò9ä!Ÿw™wŸ|™|Ð,IÐVtÓu�Ø;×DÑDÐE^Ó_’
Ø ¨Ó-´"·'±'·.±.Ü—‘—‘Ð:Ä|ÔTeÐgnÓGoÓp÷3ò 3ô  "Ÿw™wŸ|™|Ø1¼lÔK\Ð^eÓ>fó ’ð !¨Ó-´"·'±'·.±.Ü—‘—‘Ð:Ä|ÔTkÐmtÓGuÓv÷3ò 3ô  "Ÿw™wŸ|™|Ø1¼lÔKbÐdkÓ>ló �ð "’
ß$¬¯©¯©Ü—‘—‘Ð:Ä|ÔT`ÐbiÓGjÓk÷*ò *ô  "Ÿw™wŸ|™|Ø1¼lÌ<ÐY`Ó>aó ’÷ %¬¯©¯©Ü—‘—‘Ð:Ä|ÔTfÐhoÓGpÓq÷*ò *ô  "Ÿw™wŸ|™|Ø1¼lÔK]Ð_fÓ>gó �ð "’
ß ÜØ*¬<Ô8IÈ7Ó+SÐ*Tð UØ5Ð6°að9óð ô
 Ø*¬<Ô8IÈ7Ó+SÐ*TÐTYÔZfÔgsÐu|ÓZ}ÐY~ð +Ø+HÐ*IÈðLóð ô �W‰W�^‰^œBŸG™GŸL™L¨ÓR×SÒSØ8ˆLØŠHð .Ò9Ø9�Ø;×DÑDÐE^Ó_‘
Ø ¨Ó-Ü'Ô(9¸7ÓC‘ä'¬°gÓ>�ð ˜HØ˜7Ø˜Ø˜YØ"Ð$4ðˆOð ! .Ø˜jØ˜YØ2°EØ7¸Ø Ø˜jð	"ð "ð	"Ðô $Ó%Ô%÷ $ð $ä,Ð-MÓNÔN÷$ð $ð 'Ô&÷$ð $ð  ‘Oð ðYô )4Ð4QÐS[Ñ(rÐ_qÑ(rÐ%ð )Ò0°XÄÔN_ÐahÓAiÔ5iä,7Ø5Ü$Ô%<¸gÓFñ-ð -ñ-Ð)ð
 -Ò8Ø%)™
ß(Ø# vÔ-×2BÜJYØ =ñKØASñKÑGÐ1°8¸Zð :BÐ*¨:Ñ6Ø0Ò8Ü")Ø#@Ð"Að B$Ü$0Ô1BÀGÓ$LÐ#MÈTÔR^Ô_vÐxó  SAð  RBð Bzð!zó#ð ð 9ô $0´¸gÓ#F˜Ü0;Ø9¸8ñ1ØGYñ1Ð-ð
 )Ò0°XÄÌlÐ\cÓAdÔ5dä,7Ø5Ü$Ô%7¸ÓAñ-ð -ñ-Ð)ð
 -Ò8Ø%)˜
ð )Ò4ßCMÕ(?ÔSdÐ%à ¤\Ô3EÐ$FÔFÜ (Ð)FÐHYÑ mÐ]l× mÐ mß,äÜ#2Ø"?Ð!AØ$EÀtÐ#bÐOaÐ#bØ!9ô	÷
  ™%œ'øð
 Ò*¬xØ5´|ñ0ØGV÷0ð 0ô &Ø<Ð=ð > Ü ,¬\¸7Ó CÐDð E Ø '˜yÐ(cðeóð ô &Ø<Ð=ð > Ü ,¬\¸7Ó CÐDÀDÌÔVgÐipÓIqÐHrÐrsðuóð ÷$ Ü�K‰KÐ/°¨~Ð>Ô?Ø$0Ñ!ä�K‰KÐ/°¨z¸ÐI^ÐH_Ð`Õaç	ä�7‰7�>‰>˜)×$Ò$Ø$-Ñ!ð
 ˜YØ  .Ø˜7Ø"Ð$4Ø˜Ø˜jØ˜HØ˜YØ2°EØ7¸Ø ð"Ðô %0Ð0MÑ$oÐ\nÑ$oÐ!ð ÐßÜ-GØ)Ø!ØØ)ØØ-ØØ!ØØØ$Ø!ô.
Ñ*ÐÐ*ð" Ð-Ð-Ð-ð 7TÒ6_Ð1Ñ2ÐeiÐàÐ-Ð-Ð-øô} ô ð Üô äØ0Ð1NÐ0Oð P9à9VÐ8Wð X:Ü:FÄ|ÐU\Ó:]Ð9^Ð^_ðaóð
 ðûðús?   ÓA]- Ô]- Ô,B,]- ×7]- Ø&]- Ø8A-]- Ý-^,Þ^,Þ$^'Þ'^,c                óJ  € V ^8„  d   QhR\         \        P                  ,          \        ,          R,          R\        \         ,          R,          R\
        R\        R,          R\        R,          R\        R\        R,          R	\        \
        \        P                  3,          /# )
rŒ   r¦   NrÙ  Úconfigr¢   rñ   rª   r˜   r�   )	r­   r¯   r¦   r®   r°   r    r�   rR   rg  )r�   s   "r’   r“   r“   1  sš   € ÷ Pñ PÜ”—‘ÕœtÕ# dÕ*ðPäœ3•i $Õ&ðPô ðPô ˜T•kð	Pô
 �t•ðPô ðPô  Õ$ðPô ÔœUŸ[™[Ð(Õ)ñPr•   c                óà  € VRJpV Ee‹   \        V \        4      '       Ed?   V R8X  dÝ   \        VR4      '       d5   VP                  e'   VP                  p \        P                  RV  R24       MÈV'       d   RV9   d   VR,          p McVe   \        V4      p MSVe0   V^ ,          P                  R4      '       d   \        P                  p M \        V^ ,          RVR7      p\        V4      p \        P                  R	V  R
24       M2\        \        V 4      '       d   \        \        V 4      p M\        R4      h\        V \        4      '       d   \        \        V 4      MT p MJ\        V \        \        P                  34      '       g   \        RV  24      hM\        P                  ! 4       p Ve   VP                  V 4      p \        V \        4      '       dh   V P!                  R\        P                  ! 4       4      p\        V\        4      '       d   \        \        V4      MTp\        P#                  RV R24       MT pW‚n        VP$                   F  p	\        W)4      ;p
f   K  WŠn        K  	  W(3# )aš  Find the correct `dtype` to use based on provided arguments. Also update the `config` based on the
inferred dtype. We do the following:
1. If dtype is "auto", we try to read the config, else auto-detect dtype from the loaded state_dict, by checking
its first weights entry that is of a floating type - we assume all floating dtype weights are of the same dtype
2. Else, use the dtype provided as a dict or str
NÚautor¦   zWill use dtype=z$ as defined in model's config objectz.ggufr0  )r,  rª   zTSince the `dtype` attribute can't be found in model's config object, will use dtype=z  as derived from model's weightsze`dtype` provided as a `str` can only be `'auto'`, or a string representation of a valid `torch.dtype`z¨`dtype` can be one of: `torch.dtype`, `'auto'`, a string of a valid `torch.dtype` or a `dict` with valid `dtype` for each sub-config in composite configs, but received rÀ  zÅUsing different dtypes per module is deprecated and will be removed in future versions Setting different dtypes per backbone model might cause device errors downstream, therefore setting the dtype=z for all modules.)r‡  r­   rÀ   r¦   rÎ  rÏ  ró   r6  r¯   rî   rJ  r[  rÛ   r®   rÜ   Úupdate_dtyperÍ   Úwarning_onceÚsub_configs)r¦   rÙ  rÜ  r¢   rñ   rª   r˜   rÐ  Ú
main_dtypeÚsub_config_keyÚ
sub_configs   &&&&&&&    r’   Ú
_get_dtyperå  1  s2  € ð "¨Ð-€JàÓÜ�eœS×!Ó!Ø˜ŒÜ˜6 7×+Ò+°·±Ò0HØ"ŸL™L�EÜ—K‘K /°%°Ð8\Ð ]Õ^ç! gÐ1AÔ&AØ 0°Õ 9™Ø#Ò/Ü 4°ZÓ @™Ø)Ò5Ð:JÈ1Õ:M×:VÑ:VÐW^×:_Ò:_Ü %§¡™ä%4Ø,¨QÕ/¸fÐS_ô&˜
ô !5°ZÓ @˜Ü—K‘Kð*Ø*/¨Ð0PðRõô œ ×&Ò&Ü¤ uÓ-‘ä Ø{óð ô
 .8¸¼s×-CÒ-C”GœE 5Ô)È‰EÜ˜E¤D¬%¯+©+Ð#6×7Ò7ÜðJØJOÈðRóð ð 8ô ×'Ò'Ó)ˆàÒØ×)Ñ)¨%Ó0ˆô �%œ×ÒØ—Y‘Y˜r¤5×#:Ò#:Ó#<Ó=ˆ
Ü3=¸jÌ#×3NÒ3N”WœU JÔ/ÐT^ˆ
ä×Ñð!à!+ Ð,=ð?õ	
ð ˆ
ð „LØ ×,Ô,ˆÜ! &Ó9Ð9ˆJÔFØ)Öñ -ð ÐÐr•   c                   óª   a € ] tR tRt o Rt]V 3R lR l4       t]V 3R lR l4       tV 3R lR lt]	R	 4       t
RV 3R lR lltRV 3R lR lltRtV tR
# )ÚModuleUtilsMixini„  z@
A few utilities for `torch.nn.Modules`, to be used as a mixin.
c                ó8   <€ V ^8„  d   QhRRRS[ P                  /# ©rŒ   rš   r„   r�   )r¯   rä   )r�   r‘   s   "€r’   r“   ÚModuleUtilsMixin.__annotate__Š  s$   ø€ ÷ Añ AÐ&ð A©5¯<©<ñ Ar•   c                óB   € \        R V P                  4        4       4      # )zu
`torch.device`: The device on which the module is (assuming that all the module parameters are on the same
device).
c              3   ó8   "  € T F  qP                   x € K  	  R # 5ir—   ©rä   ©r  Úparams   & r’   r  Ú*ModuleUtilsMixin.device.<locals>.<genexpr>�  s   é € Ð@Ñ.? U—L–LÓ.?ùs   ‚©rï   Ú
parametersr™   s   &r’   rä   ÚModuleUtilsMixin.device‰  s   € ô Ñ@¨d¯o©oÔ.?Ó@Ó@Ð@r•   c                ó8   <€ V ^8„  d   QhRRRS[ P                  /# ré  )r¯   r¦   )r�   r‘   s   "€r’   r“   rê  ’  s$   ø€ ÷ ]ñ ]Ð%ð ]©%¯+©+ñ ]r•   c                óB   € \        R V P                  4        4       4      # )zg
`torch.dtype`: The dtype of the module (assuming that all the module parameters have the same dtype).
c              3   óh   "  € T F(  qP                  4       '       g   K  VP                  x € K*  	  R # 5ir—   )rÚ   r¦   rî  s   & r’   r  Ú)ModuleUtilsMixin.dtype.<locals>.<genexpr>–  s!   é € Ð\Ñ-> E×BYÑBY×B[”K�E—K–KÓ->ùs   ‚2ž2rñ  r™   s   &r’   r¦   ÚModuleUtilsMixin.dtype‘  s   € ô
 Ñ\¨T¯_©_Ô->Ó\Ó\Ð\r•   c                ó*   <€ V ^8„  d   QhRRRS[ RS[ /# )rŒ   rš   r„   Úencoder_attention_maskr�   )r   )r�   r‘   s   "€r’   r“   rê  ˜  s$   ø€ ÷ /ñ /Ð$5ð /Évð /ÑZ`ñ /r•   c                óP  € \         P                  R4       VP                  4       ^8X  d
   VR,          pVP                  4       ^8X  d
   VR,          pXP                  V P                  R7      pRV,
          \
        P                  ! V P                  4      P                  ,          pV# )z¸
Invert an attention mask (e.g., switches 0. and 1.).

Args:
    encoder_attention_mask (`torch.Tensor`): An attention mask.

Returns:
    `torch.Tensor`: The inverted attention mask.
z¡Detected the usage of `invert_attention_mask`: This function is deprecated and will be removed in v5.12.0. Please use the new API in `transformers.masking_utils`©r¦   ç      ð?©ºNNNNrÿ  rÿ  ©rÿ  NNrÿ  )rÎ  rà  Údimr:  r¦   r¯   ÚfinfoÚmin)rš   rú  Úencoder_extended_attention_masks   && r’   Úinvert_attention_maskÚ&ModuleUtilsMixin.invert_attention_mask˜  s¡   € ô 	×ÑðEô	
ð
 "×%Ñ%Ó'¨1Ô,Ø.DÀ]Õ.SÐ+Ø!×%Ñ%Ó'¨1Ô,Ø.DÐEUÕ.VÐ+ð +J×*LÑ*LÐSW×S]ÑS]Ð*LÓ*^Ð'Ø+.Ð1PÕ+PÔTY×T_ÒT_Ð`d×`jÑ`jÓTk×ToÑToÕ*oÐ'à.Ð.r•   c                óH  € \         P                  R 4       VP                  pV w  r4\        P                  ! WBR7      pVR,          P                  W4^4      VR,          8*  pVP                  VP                  4      pVP                  ^,          VP                  ^,          8  dh   VP                  ^,          VP                  ^,          ,
          p\        P                  ! \        P                  ! W4V3W&P                  R7      V.RR7      pVR,          VR,          ,          pV# )	z¶Detected the usage of `create_extended_attention_mask_for_decoder`: This function is deprecated and will be removed in v5.12.0. Please use the new API in `transformers.masking_utils`rí  ©rä   r¦   ©Úaxis)NNrÿ  )Nrÿ  NrM  rþ  r   )rÎ  rà  rä   r¯   ÚarangeÚrepeatr:  r¦   ÚshapeÚcatÚones)	Úinput_shapeÚattention_maskrä   Ú
batch_sizeÚ
seq_lengthÚseq_idsÚcausal_maskÚprefix_seq_lenÚextended_attention_masks	   &&       r’   Ú*create_extended_attention_mask_for_decoderÚ;ModuleUtilsMixin.create_extended_attention_mask_for_decoder³  s  € ä×ÑðEô	
ð
  ×&Ñ&ˆØ!,Ñˆ
Ü—,’,˜zÔ9ˆØ˜mÕ,×3Ñ3°JÈAÓNÐRYÐZgÕRhÑhˆà!—n‘n ^×%9Ñ%9Ó:ˆà×Ñ˜QÕ .×"6Ñ"6°qÕ"9Ô9Ø+×1Ñ1°!Õ4°{×7HÑ7HÈÕ7KÕKˆNÜŸ)š)ä—J’J 
¸ÐGÐPV×^oÑ^oÔpØðð ôˆKð #.¨mÕ"<¸~ÐN^Õ?_Õ"_ÐØ&Ð&r•   Nc          
      ól   <€ V ^8„  d   QhRRRS[ RS[S[R3,          RS[P                  R,          RS[ /# )	rŒ   rš   r„   r  r  .r¦   Nr�   )r   rg  rÅ   r¯   r¦   )r�   r‘   s   "€r’   r“   rê  Î  sL   ø€ ÷ 4'ñ 4'Øð4'áð4'ñ ™3 ˜8•_ð4'ñ �{‰{˜TÕ!ð	4'ñ
 
ñ4'r•   c                óê  € \         P                  R4       Vf   V P                  pVP                  4       ^8X  d   VR	,          pMnVP                  4       ^8X  d>   \	        V P
                  RR4      '       d   \        P                  W!4      pM&VR
,          pM\        RV RVP                   R24      hVP                  VR7      pRV,
          \        P                  ! V4      P                  ,          pV# )aš  
Makes broadcastable attention and causal masks so that future and masked tokens are ignored.

Arguments:
    attention_mask (`torch.Tensor`):
        Mask with ones indicating tokens to attend to, zeros for tokens to ignore.
    input_shape (`tuple[int]`):
        The shape of the input to the model.

Returns:
    `torch.Tensor` The extended attention mask, with a the same dtype as `attention_mask.dtype`.
z§Detected the usage of `get_extended_attention_mask`: This function is deprecated and will be removed in v5.12.0. Please use the new API in `transformers.masking_utils`NÚ
is_decoderz!Wrong shape for input_ids (shape z) or attention_mask (shape Ú)rü  rý  rþ  r   )rÎ  rà  r¦   r  r[  rÜ  rç  r  rÛ   r  r:  r¯   r  r  )rš   r  r  r¦   r  s   &&&& r’   Úget_extended_attention_maskÚ,ModuleUtilsMixin.get_extended_attention_maskÎ  sô   € ô$ 	×ÑðEô	
ð
 Š=Ø—J‘JˆEð ×ÑÓ 1Ô$Ø&4°]Õ&CÑ#Ø×ÑÓ! QÔ&ô �t—{‘{ L°$×7Ò7Ü*:×*eÑ*eØó+Ñ'ð +9Ð9IÕ*JÑ'äØ3°K°=Ð@[Ð\j×\pÑ\pÐ[qÐqrÐsóð ð #:×"<Ñ"<À5Ð"<Ó"IÐØ#&Ð)@Õ#@ÄEÇKÂKÐPUÓDV×DZÑDZÕ"ZÐØ&Ð&r•   c                ó0   <€ V ^8„  d   QhRRRS[ RS[ RS[/# )rŒ   rš   r„   Úonly_trainableÚexclude_embeddingsr�   )r�   rÅ   )r�   r‘   s   "€r’   r“   rê    s,   ø€ ÷ *ñ *Ð.ð *Áð *Ñbfð *Ñsvñ *r•   c                óä  € V'       dI   V P                  4        UUu. uF,  w  r4\        V\        P                  4      '       g   K'  V R2NK.  	  ppp\	        V RR4      pV'       d   ^ RIp^ pV P                  4        Fê  w  r9V'       d
   VX9   d   K  V	P                  '       g   V'       d   K2  V'       d›   \        V	XP                  P                  4      '       du   \        V	R4      '       d   V	P                  4       p
M+\        V	R4      '       d   V	P                  P                  p
M^p
W‰P                  4       ^,          V
,          ,          pKÔ  W‰P                  4       ,          pKì  	  V# u uppi )a¡  
Get number of (optionally, trainable or non-embeddings) parameters in the module.

Args:
    only_trainable (`bool`, *optional*, defaults to `False`):
        Whether or not to return only the number of trainable parameters

    exclude_embeddings (`bool`, *optional*, defaults to `False`):
        Whether or not to return only the number of non-embeddings parameters

Returns:
    `int`: The number of parameters.
z.weightÚis_loaded_in_4bitFNrQ  Úquant_storage)rZ  r‡  r   Ú	Embeddingr[  ÚbitsandbytesÚnamed_parametersr£  Ú
Params4bitrÀ   rQ  r%  ÚitemsizeÚnumel)rš   r!  r"  r^  Úmodule_typeÚembedding_param_namesr$  ÚbnbÚtotal_paramsrï  Ú	num_bytess   &&&        r’   Únum_parametersÚModuleUtilsMixin.num_parameters  s(  € ÷ à:>×:LÑ:LÔ:Nô%Ù:NÑ%6 TÔR\Ð]hÔjl×jvÑjv×RwÔ �4�&˜Ó Ñ:Nð "ñ %ô $ DÐ*=¸uÓEÐßÛ&àˆØ×0Ñ0Ö2‰KˆDß! dÐ.CÔ&CÙØ×"×"Ð"¯.©.÷ %¬°E¸3¿6¹6×;LÑ;L×)MÒ)MÜ˜u n×5Ò5Ø$)×$6Ñ$6Ó$8™	Ü  ¨×8Ò8Ø$)×$7Ñ$7×$@Ñ$@™	à$%˜	Ø §K¡K£M°AÕ$5¸	Õ$AÕA’Là §K¡K£MÕ1’Lñ 3ð" Ðùó5%s   œ$E,Á	E,r±   r—   ©FF)r²   r³   r´   rµ   r¶   r·   rä   r¦   r  Ústaticmethodr  r  r1  r¹   rº   r»   s   @r’   rç  rç  „  sn   ø‡ € ñð ÷Aó ðAð ÷]ó ð]÷/ð /ð6 ñ'ó ð'÷44'ò 4'÷l*÷ *ð *r•   rç  c                   óX   a € ] tR tRt o RtRtV 3R lR ltV 3R lR ltR tR	 t	R
t
V tR# )ÚEmbeddingAccessMixini1  zÀ
Base utilities to regroup getters and setters for embeddings.
Introduces the `input_layer_embed` attribute, which indicates
where the input embeddings come from and where they
should be set.
Úembed_tokensc                ó4   <€ V ^8„  d   QhRS[ P                  /# r‹   ©r   rV  )r�   r‘   s   "€r’   r“   Ú!EmbeddingAccessMixin.__annotate__;  s   ø€ ÷ %
ñ %
¡b§i¡iñ %
r•   c                ó  € \        V RR4      p\        WR4      ;pe   V# \        V RR4      pVe   \        W14      '       d   \        W14      # \        V RR4      pVe   \        WA4      '       d   \        WA4      # \        V RR4      pVe(   WPJd#   \        VR4      '       d   VP                  4       # \        V RR4      pVe(   W`Jd#   \        VR4      '       d   VP                  4       # \        R	V P                  P
                   R
24      h)zv
Returns the model's input embeddings.

Returns:
    `nn.Module`: A torch module mapping vocabulary to hidden states.
Ú_input_embed_layerr7  NÚ
embeddingsr~  Úlanguage_modelÚget_input_embeddingsÚ
base_modelu.   `get_input_embeddings` not autoâ€‘handled for ú"; please override in the subclass.)r[  rÀ   r?  ÚNotImplementedErrorÚ	__class__r²   )rš   r^  Údefault_embeddingr=  r~  r>  r@  s   &      r’   r?  Ú)EmbeddingAccessMixin.get_input_embeddings;  s  € ô �tÐ1°>ÓBˆô ")¨°TÓ!:Ð:ÐÒGØ$Ð$ä˜T <°Ó6ˆ
ØÒ!¤g¨j×&?Ò&?Ü˜:Ó,Ð,ä˜˜g tÓ,ˆØÒ¤¨×!5Ò!5Ü˜5Ó'Ð'ä  Ð'7¸Ó>ˆàÒ&ØÓ*Ü˜Ð(>×?Ò?à!×6Ñ6Ó8Ð8ô ˜T <°Ó6ˆ
ØÒ! jÓ&<ÄÈÐUk×AlÒAlØ×2Ñ2Ó4Ð4ä!Ø<¸T¿^¹^×=TÑ=TÐ<UÐUwÐxó
ð 	
r•   c                ó4   <€ V ^8„  d   QhRS[ P                  /# )rŒ   Úvaluer9  )r�   r‘   s   "€r’   r“   r:  b  s   ø€ ÷ 'ñ '©"¯)©)ñ 'r•   c                óL  € \        V RR4      p\        W4      '       d   \        WV4       R# \        V RR4      ;pe    \        W24      '       d   \        W2V4       R# \        V RR4      ;pe    \        WB4      '       d   \        WBV4       R# \        V RR4      ;pe+   WPJd&   \        VR4      '       d   VP                  V4       R# \        V RR4      ;pe+   W`Jd&   \        VR4      '       d   VP                  V4       R# \	        R	V P
                  P                   R
24      h)a¹  Fallback setter that handles **~70%** of models in the code-base.

Order of attempts:
1. `self.<_input_embed_layer>` (direct attribute)
2. `self.embeddings.<_input_embed_layer>` (nested embeddings for vision/audio models)
3. `self.model.<_input_embed_layer>` (encoder/decoder models)
4. delegate to the *base model* if one exists
5. otherwise raise `NotImplementedError` so subclasses still can (and
    should) override for exotic layouts.
r<  r7  r=  Nr~  r>  Úset_input_embeddingsr@  u.   `set_input_embeddings` not autoâ€‘handled for rA  )r[  rÀ   r¦  rI  rB  rC  r²   )rš   rG  r^  r=  r~  r>  r@  s   &&     r’   rI  Ú)EmbeddingAccessMixin.set_input_embeddingsb  s  € ô �tÐ1°>ÓBˆä�4×ÒÜ�D Ö&ä# D¨,¸Ó=Ð=ˆjÒJÌwÐWa×OhÒOhÜ�J eÖ,ä˜t W¨dÓ3Ð3ˆeÒ@ÄWÈU×EYÒEYÜ�E Ö'ô  ' tÐ-=¸tÓDÐDˆ^ÒQØÓ*Ü˜Ð(>×?Ò?à×/Ñ/°Ö6ô # 4¨°tÓ<Ð<ˆZÒIØÓ&Ü˜
Ð$:×;Ò;à×+Ñ+¨EÖ2ä%Ø@ÀÇÁ×AXÑAXÐ@YÐY{Ð|óð r•   c                óˆ   € \        V R 4      '       g   R#  V P                  4        V P                  #   \         d     R# i ; i)Úlm_headN)rÀ   r?  rB  rL  r™   s   &r’   Úget_output_embeddingsÚ*EmbeddingAccessMixin.get_output_embeddings‹  sE   € Ü�t˜Y×'Ò'Ùð	ð ×%Ñ%Ô'ð �|‰|Ðøô #ô 	Úð	ús   –2 ²AÁ Ac                ó:   € \        V R4      '       d	   Wn        R# R# )zU
Sets the model's output embedding, defaulting to setting new_embeddings to lm_head.
rL  N)r[  rL  )rš   Únew_embeddingss   &&r’   Úset_output_embeddingsÚ*EmbeddingAccessMixin.set_output_embeddings–  s   € ô �4˜×#Ò#Ø)ŽLñ $r•   )rL  N)r²   r³   r´   rµ   r¶   r<  r?  rI  rM  rQ  r¹   rº   r»   s   @r’   r6  r6  1  s7   ø‡ € ñð (Ð÷%
ð %
÷N'ð 'òR	÷*ð *r•   r6  c                   óð  a a€ ] tR tRt oRtRt]tRtRt	Rt
RtRtRtRtRtRtRtRtRtRtRtRtRtRtRtRtRtRtRtRtRtRt ]!]"PF                  PH                  V3R lR	 l4       4       t%]!V3R
 lR l4       t&V 3R lt'V3R lV 3R llt(R t)]!V3R lR l4       t*]!V3R lR l4       t+]*PX                  V3R lR l4       t*]+PX                  V3R lR l4       t+R¨R lt-R t.V3R lR lt/]0R 4       t1]!V3R lR l4       t2]0V3R lR  l4       t3R©V3R! lR" llt4RªV3R# lR$ llt5RªV3R% lR& llt6V3R' lR( lt7RªV3R) lR* llt8R«V3R+ lR, llt9V3R- lR. lt:RªV3R/ lR0 llt;V3R1 lR2 lt<]0V3R3 lR4 l4       t=]0V3R5 lR6 l4       t>RªV3R7 lR8 llt?V3R9 lR: lt@R; tAR< tBR¨V3R= lR> lltCR¨V3R? lR@ lltDRA tERB tF]"PŽ                  ! 4       RC 4       tHRªV3RD lRE lltI]"PŽ                  ! 4       ]JP–                  ! 4       RF 4       4       tLRªV3RG lRH lltMR¬V3RJ lRK lltNRL tOR­V3RM lRN lltPR¬RO ltQR­V3RP lRQ lltRR®V3RR lRS lltSRT tTRªV3RU lRV lltURW tVRX tWV3RY lRZ ltXV3R[ lR\ ltYR] tZR¨R^ lt[RI]\3V3R_ lR` llt]Ra t^]!V3Rb lRc l4       t_R¯V3Rd lRe llt`]a! ]bPÆ                  4      V 3Rf l4       tcR°Rg ltd]a! ]"PÊ                  PÌ                  PÎ                  4      V 3Rh l4       tg]a! ]"PÊ                  PÌ                  PÐ                  4      V 3Ri l4       thV 3Rj ltiV 3Rk ltj]0V3Rl lRm l4       tkV3Rn lRo ltlR¨V3Rp lRq lltm]0RrRRsRRtRRuRRvRRwRRxRyRzRR{RIR|RR}R/V3R~ lR ll4       tn]oR¨V3R€ lR� ll4       tp]oV3R‚ lRƒ l4       tqR«R„ ltr]0R±R… l4       tsR† tt]!R‡ 4       tu]!Rˆ 4       tv]!R‰ 4       tw]!RŠ 4       tx]xPX                  R‹ 4       txR¨RŒ lty]!V3R� lRŽ l4       tz]zPX                  V3R� lR� l4       tzV3R‘ lR’ lt{V3R“ lR” lt|]0R• 4       t}V3R– lR— lt~V3R˜ lR™ ltV3Rš lR› lt€Rœ t�V3R� lRž lt‚R²V3RŸ lR  lltƒR°V3R¡ lV 3R¢ lllt„R£ t…]0V3R¤ lR¥ l4       t†V3R¦ lt‡R§tˆVt‰V ;tŠ# )³r„   iž  am  
Base class for all models.

[`PreTrainedModel`] takes care of storing the configuration of the models and handles methods for loading,
downloading and saving models as well as a few methods common to all models to:

    - resize the input embeddings

Class attributes (overridden by derived classes):

    - **config_class** ([`PreTrainedConfig`]) -- A subclass of [`PreTrainedConfig`] to use as configuration class
      for this model architecture.
    - **base_model_prefix** (`str`) -- A string indicating the attribute associated to the base model in derived
      classes of the same architecture adding modules on top of the base model.
    - **main_input_name** (`str`) -- The name of the principal input to the model (often `input_ids` for NLP
      models, `pixel_values` for vision models and `input_values` for speech models).
    - **can_record_outputs** (dict):
NrÀ  FÚ	input_idsÚtextc                ó6   <€ V ^8„  d   QhRS[ S[S[3,          /# r‹   )r®   r­   rz   )r�   r‘   s   "€r’   r“   ÚPreTrainedModel.__annotate__ó  s   ø€ ÷ '.ñ '.¡D©©nÐ)<Õ$=ñ '.r•   c                ó.   € V P                   ;'       g    / # )aã  
 Maps output names (e.g., "attentions", "hidden_states")
 to either:
     - A module class (e.g., `LlamaDecoderLayer`), using default index conventions:
         * index=0 for "hidden_states"
         * index=1 for "attentions"
     - Or an `OutputRecorder(...)` with `target_class`, optional `index`, and `layer_name`.

 Examples:
     These two are equivalent:

 ```python
     _can_record_outputs = {
         "attentions": LlamaAttention,
         "hidden_states": LlamaDecoderLayer
     }

     _can_record_outputs = {
         "attentions": OutputRecorder(LlamaAttention, index=1),
         "hidden_states": OutputRecorder(LlamaDecoderLayer, index=0)
     }
```

 This means you can record outputs from the same class, by specifying a layer name. Before
 collecting outputs, we check that they come from this layer.

 If you have cross attention that come from `LlamaAttention` and self attention that also
 come from `LlamaAttention` but from `self_attn` you can do this:

 ```python
 class LlamaModel(PreTrainedModel):
     _can_record_outputs = {
         "attentions": OutputRecorder(LlamaAttention, index=1, layer-name="self_attn"),
         "cross_attentions": OutputRecorder(LlamaAttention, index=1, layer_name="cross_attn")
     }

```
)Ú_can_record_outputsr™   s   &r’   Úcan_record_outputsÚ"PreTrainedModel.can_record_outputsñ  s   € ðR ×'Ñ'×-Ð-¨2Ð-r•   c                óJ   <€ V ^8„  d   QhRS[ S[S[P                  3,          /# r‹   r  )r�   r‘   s   "€r’   r“   rW    s"   ø€ ÷ 9ñ 9™d¡3©¯©Ð#4Õ5ñ 9r•   c                ó:   € R\         P                  ! \        4      /# )zN
`dict[str, torch.Tensor]`: Dummy inputs to do a forward pass in the network.
rT  )r¯   rã   rW   r™   s   &r’   Údummy_inputsÚPreTrainedModel.dummy_inputs  s   € ð
 œUŸ\š\¬,Ó7Ð8Ð8r•   c                óZ  <€ \         SV `  ! R/ VB  \        P                  ! V 4      P	                  R R4      pV P
                  P	                  RR4      p\        V 4      P	                  R R4      pV P                  pVe	   W0n        R# Ve	   W n        R# Ve	   WPn        R# Ve	   W@n        R# R# )rÜ  NÚconfig_classr±   )ÚsuperÚ__init_subclass__ÚinspectÚget_annotationsrÍ   Ú__dict__r   ra  )ÚclsrÉ  Úchild_annotationÚchild_attributeÚfull_annotationÚfull_attributerC  s   &,    €r’   rc  Ú!PreTrainedModel.__init_subclass__#  s¤   ø€ Ü‰Ò!Ñ+ FÒ+ô #×2Ò2°3Ó7×;Ñ;¸HÀdÓKÐØŸ,™,×*Ñ*¨>¸4Ó@ˆô )¨Ó-×1Ñ1°(¸DÓAˆØ×)Ñ)ˆð Ò&Ø.ÖØÒ)Ø/ÖØÒ'Ø-ÖØÒ(Ø.Öñ )r•   c                ó    <€ V ^8„  d   QhRS[ /# )rŒ   rÜ  r   )r�   r‘   s   "€r’   r“   rW  ;  s   ø€ ÷ -Mñ -MÑ/ñ -Mr•   c                ó  <€ \         SV `  4        \        V\        4      '       g;   \	        R V P
                  P                   RV P
                  P                   R24      hWn        VP                  V n        RV n	        V P                  V P                  P                  R\        P                  R7      V P                  n        V P                  V P                  P                   4      V P                  n        V P%                  4       '       d"    V P&                  P)                  V4      V n        V P
                  P                  pV\.        9  d`   RRP1                  \.        4       R2p\2        P4                  ! WPP
                  P                  4      p\7        V4      ^ 8”  d   V^ ,          pMRpW@n        V P:                  \<        \?        V P
                  4      &   R#   \,         d    T P'                  4       T n         LÐi ; i)	zParameter config in `zt(config)` should be an instance of class `PreTrainedConfig`. To create a model from a pretrained model use `model = z(.from_pretrained(PRETRAINED_MODEL_NAME)`NT©Úis_init_checkÚallow_all_kernelsÚ(Ú|r  ) rb  Ú__init__r‡  r    Ú	TypeErrorrC  r²   rÜ  Úname_or_pathÚkernel_configÚ%_check_and_adjust_attn_implementationÚ_attn_implementationr,   ÚALLOW_ALL_KERNELSÚ_attn_implementation_internalÚ(_check_and_adjust_experts_implementationÚ_experts_implementationÚ _experts_implementation_internalÚcan_generateÚgeneration_config_classÚfrom_model_configÚgeneration_configrB  rI   rË  rƒ  Úfindallrí   Ú	loss_typerY  ry   r­   )rš   rÜ  ÚinputsrÉ  r„  Úloss_groupsrC  s   &&*,  €r’   rt  ÚPreTrainedModel.__init__;  s®  ø€ Ü‰ÑÔÜ˜&Ô"2×3Ò3ÜØ'¨¯©×(?Ñ(?Ð'@ð Aà ŸN™N×3Ñ3Ð4Ð4\ð^óð ð
 ŒØ"×/Ñ/ˆÔð "ˆÔð 59×4^Ñ4^Ø�K‰K×,Ñ,Øä)×;Ñ;ð	 5_ó 5
ˆ�‰Ô1ð 8<×7dÑ7dØ�K‰K×/Ñ/ó8
ˆ�‰Ô4ð ×Ñ×ÒðHØ)-×)EÑ)E×)WÑ)WÐX^Ó)_�Ô&ð
 —N‘N×+Ñ+ˆ	ØœLÔ(Ø˜cŸh™h¤|Ó4Ð5°QÐ7ˆKÜŸ
š
 ;·±×0GÑ0GÓHˆIÜ�9‹~ Ô!Ø% a�L‘	à �	Ø"Œà48×4LÑ4LÔœS §¡Ó0Ó1øô 'ô HØ)-×)EÑ)EÓ)G�Ö&ðHús   Ä G Ç G?Ç>G?c                óÂ
  € / / / uV n         V n        V n        V P                  V J dÊ   V P                  P
                  e%   V P                  P
                  P                  4       M/ V n        V P                  P                  e%   V P                  P                  P                  4       M/ V n         V P                  P                  e%   V P                  P                  P                  4       M/ V n        V P                  RR7      V n
        \        V P                  ;'       g    . 4      V n        \        V P                  ;'       g    . 4      V n        \        V P                  ;'       g    . 4      V n        \        V P                  ;'       g    . 4      V n        \        V P                   ;'       g    . 4      V n        \        V P"                  ;'       g    . 4      V n        \        V P$                  ;'       g    . 4      V n        V P'                  4        EF
  w  r\)        VRR4      ;p'       dR   V P                  P+                  VP                  4       P-                  4        UUu/ uF  w  rEV RV 2VbK  	  upp4       \)        VRR4      ;p'       dR   V P                   P+                  VP                  4       P-                  4        UUu/ uF  w  rEV RV 2VbK  	  upp4       \)        VRR4      ;p'       dR   V P                  P+                  VP                  4       P-                  4        UUu/ uF  w  rEV RV 2VbK  	  upp4       \)        VRR4      ;p'       dW   V P                  P+                  VP                  4       P-                  4        UUu/ uF  w  rEV RV 2V RV 2bK  	  upp4       \)        VR	R4      ;p'       d   V P                  P+                  V4       \)        VR
R4      ;p'       d   V P                  P+                  V4       \)        VRR4      ;p	'       d   V P                  P+                  V	4       \)        VRR4      ;p
'       d   V P                  P+                  V
4       \)        VRR4      ;p'       d   V P                   P+                  V4       \)        VRR4      ;p'       d   V P"                  P+                  V4       \)        VRR4      ;p'       g   EKÛ  V P$                  P+                  V Uu0 uF	  qA RV 2kK  	  up4       EK  	  V P/                  4        V P1                  4        R# u uppi u uppi u uppi u uppi u upi )a§  
A method executed at the end of each Transformer model initialization, to execute code that needs the model's
modules properly initialized (such as weight initialization).
It is also used to obtain all correct static properties (parallelism plans, tied_weights_keys, _keep_in_fp32_modules, etc)
correctly in the case of composite models (that is, the top level model should know about those properties from its children).
NF©Úall_submodelsÚ_ep_planrY  Ú_tp_planÚ_pp_planÚall_tied_weights_keysÚ_keep_in_fp32_modulesÚ_keep_in_fp32_modules_strictÚ_no_split_modulesÚ_skip_keys_device_placementÚ"_keys_to_ignore_on_load_unexpectedÚ_keys_to_ignore_on_load_missingÚ_keys_to_ignore_on_save)rŒ  r‹  r�  r@  rÜ  Úbase_model_pp_planÚcopyÚbase_model_tp_planÚbase_model_ep_planÚget_expanded_tied_weights_keysrŽ  rf  r�  r�  r‘  r’  r“  r”  r•  Únamed_childrenr[  Úupdater9  Úinit_weightsÚ._backward_compatibility_gradient_checkpointing)rš   r^  rU  ÚplanrD  rE  Ú	tied_keysÚ	keep_fp32Úkeep_fp32_strictÚno_splitÚ	skip_keysÚignore_unexpectedÚignore_missingÚignore_saves   &             r’   Ú	post_initÚPreTrainedModel.post_initj  s}  € ð 79¸"¸bÐ3ˆŒ�t”} d¤mà�?‰?˜dÓ"ØEIÇ[Á[×EcÑEcÒEo˜DŸK™K×:Ñ:×?Ñ?ÔAÐuwˆDŒMØEIÇ[Á[×EcÑEcÒEo˜DŸK™K×:Ñ:×?Ñ?ÔAÐuwˆDŒMØEIÇ[Á[×EcÑEcÒEo˜DŸK™K×:Ñ:×?Ñ?ÔAÐuwˆDŒMà%)×%HÑ%HÐW\Ð%HÓ%]ˆÔ"ä%(¨×)CÑ)C×)IÐ)IÀrÓ%JˆÔ"Ü,/°×0QÑ0Q×0WÐ0WÐUWÓ,XˆÔ)ä!$ T×%;Ñ%;×%AÐ%A¸rÓ!BˆÔÜ+.¨t×/OÑ/O×/UÐ/UÐSUÓ+VˆÔ(ä25°d×6]Ñ6]×6cÐ6cÐacÓ2dˆÔ/Ü/2°4×3WÑ3W×3]Ð3]Ð[]Ó/^ˆÔ,Ü'*¨4×+GÑ+G×+MÐ+MÈ2Ó'NˆÔ$ð !×/Ñ/×1‰LˆDä˜v z°4Ó8Ð8ˆtÖ8Ø—‘×$Ñ$À4Ç9Á9Ã;×CTÑCTÔCVÔ%WÑCV¹4¸1¨¨¨a°¨s m°QÒ&6ÑCVÒ%WÔXÜ˜v z°4Ó8Ð8ˆtÖ8Ø—‘×$Ñ$À4Ç9Á9Ã;×CTÑCTÔCVÔ%WÑCV¹4¸1¨¨¨a°¨s m°QÒ&6ÑCVÒ%WÔXÜ˜v z°4Ó8Ð8ˆtÖ8Ø—‘×$Ñ$À4Ç9Á9Ã;×CTÑCTÔCVÔ%WÑCV¹4¸1¨¨¨a°¨s m°QÒ&6ÑCVÒ%WÔXä# FÐ,CÀTÓJÐJˆyÖJØ×*Ñ*×1Ñ1Ð\e×\jÑ\jÓ\l×\rÑ\rÔ\tÔ2uÑ\tÑTXÐTU°d°V¸1¸Q¸C°=ÀTÀFÈ!ÈAÈ3À-Ò3OÑ\tÒ2uÔvä# FÐ,CÀTÓJÐJˆyÖJØ×*Ñ*×1Ñ1°)Ô<Ü#*¨6Ð3QÐSWÓ#XÐXÐÖXØ×1Ñ1×8Ñ8Ð9IÔJä" 6Ð+>ÀÓEÐEˆxÖEØ×&Ñ&×-Ñ-¨hÔ7Ü# FÐ,IÈ4ÓPÐPˆyÖPØ×0Ñ0×7Ñ7¸	ÔBô %,¨FÐ4XÐZ^Ó$_Ð_Ð Ö_Ø×7Ñ7×>Ñ>Ð?PÔQÜ!(¨Ð1RÐTXÓ!YÐYˆ~ÖYØ×4Ñ4×;Ñ;¸NÔKä% fÐ.GÈÓNÐNˆ{×NÑNØ×,Ñ,×3Ñ3ÉKÓ4XÉKÀq°v¸Q¸q¸c³]ÉKÑ4X×Yñ; 2ð@ 	×ÑÔØ×;Ñ;Ö=ùó= &Xùã%Wùã%Wùó 3vùò& 5Ys   É$UË
U
Ì0UÎUÔU
c                ó6   <€ V ^8„  d   QhRS[ S[S[3,          /# r‹   ©r®   r­   )r�   r‘   s   "€r’   r“   rW  ¬  s   ø€ ÷ ñ ™™c¡3˜h�ñ r•   c                ó¶   € \        V P                  R4      '       d3   V P                  P                  P                  '       d   V P                  # V P
                  # )z*
The full tp plan for the model's modules
Údistributed_config)rÀ   rÜ  r­  Úenable_expert_parallelr‹  rŒ  r™   s   &r’   Útp_planÚPreTrainedModel.tp_plan«  s?   € ô
 �4—;‘;Ð 4×5Ò5¸$¿+¹+×:XÑ:X×:o×:oÐ:oØ—=‘=Ð Ø�}‰}Ðr•   c                óL   <€ V ^8„  d   QhRS[ S[S[S[S[3,          3,          /# r‹   ©r®   r­   rg  )r�   r‘   s   "€r’   r“   rW  µ  s&   ø€ ÷ ñ ™™c¡5©©c¨¥?Ð2Õ3ñ r•   c                ó   € V P                   # r—   )r�  r™   s   &r’   Úpp_planÚPreTrainedModel.pp_plan´  s   € à�}‰}Ðr•   c                óD   <€ V ^8„  d   QhRS[ S[S[3,          R,          /# ©rŒ   rŸ  Nr«  )r�   r‘   s   "€r’   r“   rW  ¹  s!   ø€ ÷ !ñ !™D¡¡c �N¨TÕ1ñ !r•   c                óX  € Vf
   / V n         R # \        V\        4      '       g   \        R4      hVP	                  4        F@  w  r#V\
        9  g   K  \        RV RV R\        \
        P                  ! 4       4       24      h	  V P                  4        UUu. uF  w  rEVNK	  	  pppVP                  4        Fd  pVP                  RR4      pRpV F#  p	\        P                  ! Wy4      '       g   K!  Rp M	  V'       d   KJ  \        P                  ! R	V R
24       Kf  	  Wn         R # u uppi )Nz&Can only set a dictionary as `tp_plan`z#Unsupported tensor parallel style 'z' for layer 'z'. Supported styles are Ú*z\d+FTzLayer pattern 'z�' does not match any parameters in the model. This rule may not be applied during tensor parallelization, or may lead to dimension mismatches)rŒ  r‡  r®   rÛ   r9  rB   r°   r;  r(  Úreplacerƒ  ÚmatchÚwarningsÚwarn)
rš   rŸ  Úlayer_patternÚparallel_styler^  rp  Úmodel_param_namesÚregex_patternÚpattern_matchedr¡  s
   &&        r’   r¯  r°  ¸  s%  € àŠ<ØˆDŒMÙÜ˜$¤×%Ò%ÜÐEÓFÐFð .2¯Z©Z®\Ñ)ˆMØÔ%8Ö8Ü Ø9¸.Ð9IÈÐWdÐVeð f,Ü,0Ô1D×1IÒ1IÓ1KÓ,LÐ+MðOóð ñ .:ð 26×1FÑ1FÔ1HÔIÑ1H¡g d›TÑ1HÐÑIØ!ŸY™Yž[ˆMà)×1Ñ1°#°vÓ>ˆMØ#ˆOÛ/�
Ü—8’8˜M×6Ô6Ø&*�OÙñ 0÷ #‘?Ü—’Ø% m _ð 5dð döñ )ð Žùó! Js   ÂD&c                óZ   <€ V ^8„  d   QhRS[ S[S[S[S[3,          3,          R,          /# r·  r²  )r�   r‘   s   "€r’   r“   rW  Ý  s+   ø€ ÷ ñ ™D¡¡e©C±¨H¥oÐ!5Õ6¸Õ=ñ r•   c                ón   € Vf
   / V n         R # \        V\        4      '       g   \        R4      hWn         R # )Nz&Can only set a dictionary as `pp_plan`)r�  r‡  r®   rÛ   )rš   rŸ  s   &&r’   r´  rµ  Ü  s/   € àŠ<ØˆDŒMÙÜ˜$¤×%Ò%ÜÐEÓFÐFàŽr•   c                ó^   € \        V RR4      pVf   \        R4      hVP                  WR7      # )zv
Potentially dequantize the model in case it has been quantized by a quantization method that support
dequantization.
r˜   Nz?You need to first quantize your model in order to dequantize itrü  )r[  rÛ   Ú
dequantize)rš   r¦   r˜   s   && r’   rÆ  ÚPreTrainedModel.dequantizeæ  s8   € ô
 ˜t ^°TÓ:ˆàÒÜÐ^Ó_Ð_à×&Ñ& tÐ&Ó9Ð9r•   c                ó¸   € V P                   '       dH   \        V P                  R R4      '       d)   V P                  4        \	        V P                  R 4       R# R# R# )Úgradient_checkpointingFN)Úsupports_gradient_checkpointingr[  rÜ  Úgradient_checkpointing_enableÚdelattrr™   s   &r’   rž  Ú>PreTrainedModel._backward_compatibility_gradient_checkpointingò  sF   € Ø×/×/Ð/´G¸D¿K¹KÐIaÐch×4iÒ4iØ×.Ñ.Ô0ä�D—K‘KÐ!9Ö:ñ 5jÑ/r•   c                óD   <€ V ^8„  d   QhRS[ S[,          S[,          RR/# )rŒ   Útagsr�   N)r°   r­   )r�   r‘   s   "€r’   r“   rW  ø  s#   ø€ ÷ ,ñ ,¡4©¥9©s¥?ð ,°tñ ,r•   c                óÎ   € \        V\        4      '       d   V.pV P                  f   . V n        V F0  pW P                  9  g   K  V P                  P                  V4       K2  	  R# )aì  
Add custom tags into the model that gets pushed to the Hugging Face Hub. Will
not overwrite existing tags in the model.

Args:
    tags (`Union[list[str], str]`):
        The desired tags to inject in the model

Examples:

```python
from transformers import AutoModel

model = AutoModel.from_pretrained("google-bert/bert-base-cased")

model.add_model_tags(["custom", "custom-bert"])

# Push the model to your namespace with the name "my-custom-bert".
model.push_to_hub("my-custom-bert")
```
N)r‡  r­   Ú
model_tagsri  )rš   rÏ  Útags   && r’   Úadd_model_tagsÚPreTrainedModel.add_model_tagsø  sO   € ô, �dœC× Ò Ø�6ˆDà�?‰?Ò"Ø ˆDŒOãˆCØŸ/™/Ö)Ø—‘×&Ñ& sÖ+ó r•   c                ó¾  € VP                  RVP                  4      pVP                  RR4      ;pe*   \        P                  R4       W1P                  8w  d   TMTp\	        V\
        4      '       d   \        \        V4      pW1n        VP                   F  p\        W4      ;pf   K  W6n        K  	  RV9   d   VP                  R4      Vn	        RV9   d   VP                  R4      Vn
        VP                  RR4      p\        4       .pVe%   VP                  \        W0P                  4      4       V'       d   VP                  \!        4       4       \#        4       ;'       d    \$        '       * ;'       d    \&        '       * p	V	'       dk   \        P)                  R	4       ^ RIp
VP-                  \.        P0                  ! 4       V
P2                  P5                  \7        4       R
7      \9        4       .4       \;        V4      ;_uu_ 4        V ! V3/ VB p\=        V4       RRR4       V	'       d   ^RIH p V! X4       VPC                  4        X#   + '       g   i     L8; i)zÂ
All context managers that the model should be initialized under go here.

Args:
    dtype (`torch.dtype`, *optional*):
        Override the default `dtype` and load the model under this dtype.
r¦   Útorch_dtypeNz1`torch_dtype` is deprecated! Use `dtype` instead!Úattn_implementationÚexperts_implementationrq  Fú@Detected DeepSpeed ZeRO-3: activating zero.init() for this model©Úconfig_dict_or_path)Úinitialize_weights_zero3)"rl  r¦   rÎ  rà  r‡  r­   r[  r¯   rá  ry  r}  rÍ   rO   ri  rà   r²   r<   r-   rÑ   rÕ   rÏ  Ú	deepspeedr\  ÚinitÚno_init_weightsÚzeroÚInitr+   rÖ   r\   rP   Úintegrations.deepspeedrÜ  Útie_weights)rg  rÜ  rÉ  r¦   rÖ  rã  rä  rq  Úinit_contextsÚneeds_zero3_initrÝ  r~  rÜ  s   &&,          r’   Ú_from_configÚPreTrainedModel._from_config  sâ  € ð —
‘
˜7 F§L¡LÓ1ˆØ!Ÿ:™: m°TÓ:Ð:ˆKÒGÜ×ÑÐ SÔTà"§l¡lÔ2‘E¸ˆEÜ�eœS×!Ò!ÜœE 5Ó)ˆEð
 ŒØ$×0Ô0ˆNÜ% fÓ=Ð=�
ÔJØ#(Ö ñ 1ð
 ! FÔ*Ø*0¯*©*Ð5JÓ*KˆFÔ'ð $ vÔ-Ø-3¯Z©ZÐ8PÓ-QˆFÔ*ð #ŸJ™JÐ':¸EÓBÐä&›Ð)ˆØÒØ× Ñ Ô!2°5¿,¹,Ó!GÔHßØ× Ñ Ô!6Ó!8Ô9ä5Ó7×hÐhÄÔ<M×hÐhÔVhÔRhÐßÜ�K‰KÐZÔ[ó à× Ñ ä×(Ò(Ó*Ø—N‘N×'Ñ'Ô<LÓ<NÐ'ÓOÜ#Ó%ðôô ˜]×+Õ+Ù˜Ñ) &Ñ)ˆEÜ" 5Ô)÷ ,÷ ÝHá$ UÔ+Ø×ÑÔàˆ÷ ,×+ús   ÈIÉI	c                ó4   <€ V ^8„  d   QhRS[ P                  /# r‹   r9  )r�   r‘   s   "€r’   r“   rW  c  s   ø€ ÷ ;ñ ;™BŸI™Iñ ;r•   c                ó.   € \        W P                  V 4      # )z0
`torch.nn.Module`: The main body of the model.
)r[  Úbase_model_prefixr™   s   &r’   r@  ÚPreTrainedModel.base_modelb  s   € ô
 �t×3Ñ3°TÓ:Ð:r•   c                ó    <€ V ^8„  d   QhRS[ /# r‹   rŽ   )r�   r‘   s   "€r’   r“   rW  j  s   ø€ ÷ $ñ $™Tñ $r•   c                óJ  € R\        V P                  4      9   d   R# V P                   FB  p\        VR4      '       g   K  R\        V4      9  g   K)  VP                  4       '       g   KA   R# 	  \        V R4      '       d#   \        P                  V P                   R24       R# )a›  
Returns whether this model can generate sequences with `.generate()`, from the `GenerationMixin`
or one with a similar interface.

Under the hood, on classes where this function returns True, some generation-specific changes are triggered:
for instance, the model instance will have a populated `generation_config` attribute.

Returns:
    `bool`: Whether this model can generate sequences with `.generate()`.
ÚGenerationMixinTr  r„   Úprepare_inputs_for_generationu6   has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From ðŸ‘‰v4.50ðŸ‘ˆ onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions.
  - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes
  - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception).
  - If you are not the owner of the model architecture class, please contact the model code owner to update it.F)r­   Ú	__bases__rÀ   r  rÎ  Úwarningr²   )rg  Úbases   & r’   r  ÚPreTrainedModel.can_generatei  s‡   € ð ¤ C§M¡MÓ 2Ô2Ùà—M”MˆDÜ˜4 ×0Ò0ÙØ ¬¨D«	Ö1°d×6GÑ6G×6IÔ6IÚñ	 "ô �3Ð7×8Ò8Ü�N‰NØ—<‘<�.ð 	! ð 	 ôñ r•   c                ó    <€ V ^8„  d   QhRS[ RS[RS[RS[S[S[S[3,          R3,          RS[S[S[S[3,          R3,          RS[ R,          /# )	rŒ   Úflash_attn_versionÚgeneral_availability_checkÚpkg_availability_checkÚsupported_devices.Úcustom_supported_devicesÚcuda_min_major_versionN)rÅ   r   rg  r­   )r�   r‘   s   "€r’   r“   rW  �  sy   ø€ ÷ @ñ @áð@ñ %-ð@ñ !)ð	@ñ
 !¡¡x± }Õ!5°sÐ!:Õ;ð@ñ #(©©h¹¨mÕ(<¸cÐ(AÕ"Bð@ñ !$ d¥
ñ@r•   c                óº  € V F*  w  rxV! 4       '       g   K  \         P                  V4        R# 	  V! 4       '       Eg   RV R2p	V! 4       '       g   \        V	 RV R24      hV^8X  d#   \        R4      '       g   \        V	 RV R24      h\	        V!  w  r«\
        ;QJ d    R	 V
 4       F  '       g   K   R
M	  RM! R	 V
 4       4      '       g   \        V	 RV RV R24      hVeq   \        4       '       d_   \        P                  P                  4       w  rÍWÆ8  d7   \        V	 RV RV R\        P                  P                  4        RV R2
4      hR# R# R# R# )a�  
Checks whether the specified Flash Attention version is supported and if not, searches for the specific reason
on why it failed - package import and/or device incompatibility issues.

Args:
    flash_attn_version (`int`):
        The requested version of Flash Attention.
    general_availability_check (`Callable`):
        Checks whether our `is_available` function detects the specific FA version. Failing reasons
        are then checked for one-by-one.
    pkg_availability_check (`Callable`):
        Checks whether the package could theoretically be detected in the environment by the init structures.
        This is not a sure-fire check as device compatibility with FA is just as important.
    supported_devices (`tuple[tuple[Callable, str]]`):
        Essentially a list (for mutable kwargs reasons a tuple) of the supported devices in the format of
        `(device_availability_check, device_name)`, i.e. a pair of the associated device's name and whether
        it is available in the environment.
    custom_supported_devices (`tuple[tuple[Callable, str]]`, *optional*, defaults to `()`):
        Essentially a list (for mutable kwargs reasons a tuple) of the custom supported devices in the format of
        `(device_availability_check, info_message)`. These custom devices have custom logic outside the torch
        ecosystem either via kernels or other packages and hence have early checks for availability.
    cuda_min_major_version (`int`, *optional*):
        The minimum major cuda version supported for this version of Flash Attention. This is mostly
        affecting more recent versions which are more specialized to the features of new hardware.
NÚFlashAttentionzG has been toggled on, but it cannot be used due to the following error:z the package for FlashAttentionz doesn't seem to be installed.z2.3.3z FlashAttentionz# requires at least version `2.3.3`.c              3   ó.   "  € T F  q! 4       x € K  	  R # 5ir—   r±   )r  Údevice_availability_checks   & r’   r  Ú;PreTrainedModel._flash_attn_import_error.<locals>.<genexpr>Æ  s   é € ÐsÑXrÐ;TÐ4×6Ð6ÓXrùó   ‚TFzT is not available on CPU. Please make sure you are on any of the supported devices: rY  z  requires compute capability >= z, but found z with compute capability z.x)
rÎ  rÏ  ÚImportErrorrr   Úzipr‹  ru   r¯   ÚcudaÚget_device_capability)rš   rõ  rö  r÷  rø  rù  rú  rþ  Úinfo_messageÚprefaceÚdevice_availability_checksÚdevice_namesÚmajorrp  s   &&&&&&&       r’   Ú_flash_attn_import_errorÚ(PreTrainedModel._flash_attn_import_error�  sß  € óF 8PÑ3Ð%Ù(×*Ô*Ü—‘˜LÔ)Úñ 8Pñ
 *×+Ó+Ø&Ð'9Ð&:ð  ;Bð  CˆGñ *×+Ò+Ü!Ø�iÐ>Ð?QÐ>RÐRpÐqóð ð $ qÔ(Ô1OÐPW×1XÒ1XÜ! W I¨_Ð=OÐ<PÐPsÐ"tÓuÐuô <?Ð@QÑ;RÑ8Ð*ß“sÑsÑXrÓs—s—s’sÑsÑXrÓs×sÒsÜ%Ø"˜) ?Ð3EÐ2Fð  G[ð  \hð  [ið  ijð  kóð ð ,Ò7Ô<S×<UÒ<UÜ$Ÿz™z×?Ñ?ÓA‘H�EØÔ5Ü)Ø&˜i Ð7IÐ6JÐJjð  lBð  kCð  COô  PU÷  PZñ  PZ÷  Ppñ  Ppó  Prð  Osð  sLð  MRð  LSð  SUð  Vóð ñ 6ñ =VÑ7ñ' ,r•   c                ó,   <€ V ^8„  d   QhRS[ RS[RS[/# )rŒ   rõ  rp  r�   )rÅ   r�   )r�   r‘   s   "€r’   r“   rW  Ò  s(   ø€ ÷ Iñ I¹3ð IÉtð IÑ`dñ Ir•   c                óp  € V P                   '       g=   \        V P                  P                   RV RV P                  P
                   R24      hVR9  d   \        RV R24      hV P                  ! R/ \        V,          B  V^8”  dQ   \        V P                  R4      '       d5   V P                  P                  ^ 8”  d   \        P                  RV R24       V P                  P                  pVf   \        P                  RV R	24       M_Ve\   V\        P                  \        P                  39  d7   \        P                  R
V RV P                  P                   RV RV R2	4       V'       g¾   \!        V P#                  4        Uu0 uF  qDP$                  kK  	  up4      p\'        V4      ^8X  d|   V^ ,          P(                  R8X  dd   Rp\        V,          R,           F2  w  rxV! 4       '       g   K  Rp\        P                  RV RV R24        M	  V'       g   \        RV R24      hR# u upi )a^  
Check the availability of Flash Attention for a given model.

Args:
    flash_attn_version (`int`):
        The requested version of Flash Attention.
    is_init_check (`bool`, *optional*):
        Whether this check is performed early, i.e. at __init__ time, or later when the model and its weights are
        fully instantiated. This is needed as we also check the devices of the weights, which are only available
        later after __init__. This allows to raise proper exceptions early before instantiating the full models
        if we know that the model does not support the requested attention.
z" does not support Flash Attention zm yet. Please request to add support where the model is hosted, on its model hub page: https://huggingface.co/zk/discussions/new or in the Transformers GitHub repo: https://github.com/huggingface/transformers/issues/newzRequested Flash Attention z which is not supported.Úattention_dropoutz*You are attempting to use Flash Attention zv with dropout. This might lead to unexpected behaviour as this is not supported on recent versions of Flash Attention.zD without specifying a dtype. This might lead to unexpected behaviourzFlash Attention zP only supports torch.float16 and torch.bfloat16 dtypes, but the current dype in z is a&  . You should run training or inference using Automatic Mixed-Precision via the `with torch.autocast(device_type='torch_device'):` decorator, or load the model with the `dtype` argument. Example: `model = AutoModel.from_pretrained("meta-llama/Llama-3.2-1B", attn_implementation="flash_attention_z", dtype=torch.float16)`râ   Frø  Tzá with a model not initialized on GPU. Please make sure to have access to a GPU and either initialise the model on a GPU by passing a device_map or initialising the model on CPU and then moving it to GPU, e.g. with `model.to('z')`.a    with a model not initialized on GPU and with no GPU available. This is not supported yet. Please make sure to have access to a GPU and either initialise the model on a GPU by passing a device_map or initialising the model on CPU and then moving it to GPU.)rŒ   é   é   r±   )Ú_supports_flash_attnrÛ   rC  r²   rÜ  Ú_name_or_pathr
  rJ   rÀ   r  rÎ  rà  r¦   r¯   Úfloat16Úbfloat16r°   rò  rä   rí   r‰  )	rš   rõ  rp  r¦   rï  Úparam_devicesÚfound_devicerþ  Údevice_names	   &&&      r’   Ú_flash_attn_can_dispatchÚ(PreTrainedModel._flash_attn_can_dispatchÒ  sv  € ð ×(×(Ð(ÜØ—>‘>×*Ñ*Ð+Ð+MÐN`ÐMað bWØW[×WbÑWb×WpÑWpÐVqð rnðnóð ð  YÔ.ÜÐ9Ð:LÐ9MÐMeÐfÓgÐgð 	×%Ò%ÑaÔ(LÐM_Õ(`Òað  Ô!Ü�t—{‘{Ð$7×8Ò8¸T¿[¹[×=ZÑ=ZÐ]^Ô=^Ü×#Ñ#Ø@ÐASÐ@Tð U~ð ~ôð —‘×!Ñ!ˆØŠ=Ü×ÑØ<Ð=OÐ<Pð  QUð  Võð Ò 5´·±ÄÇÁÐ0OÔ#OÜ×ÑØ"Ð#5Ð"6ð 7(Ø(,¯©×(?Ñ(?Ð'@ÀÀUÀGð Lmð n@ð  mAð  AYðZô÷ Ü ¸D¿O¹OÔ<MÓ!NÑ<M°5§,¤,Ñ<MÑ!NÓOˆMÜ�=Ó! QÔ&¨=¸Õ+;×+@Ñ+@ÀEÔ+IØ$�Ü>bÐcuÕ>vØ'÷?ð ?Ñ:Ð-ñ 1×2Ô2Ø'+˜Ü×+Ñ+ØHÐI[ÐH\ð ]FàFQÀ]ÐRVðXôñ
 ñ?÷ $Ü$ØDÐEWÐDXð YVð Vóð ñ ùò/ "Os   ÆH3c                ó&   <€ V ^8„  d   QhRS[ RS[ /# ©rŒ   rp  r�   rŽ   )r�   r‘   s   "€r’   r“   rW    s   ø€ ÷ ñ ±ð Áñ r•   c                óâ  € V P                   '       g#   \        V P                  P                   R24      h\        P
                  P                  eŸ   \        P                  P                  4       ^8”  d|   \
        P                  ! \        P                  4      \
        P                  ! R4      8  d?   \        P                  R4       \        P                  P                  P                  R4       R# )a  
Check the availability of SDPA for a given model.

Args:
    is_init_check (`bool`, *optional*):
        Whether this check is performed early, i.e. at __init__ time, or later when the model and its weights are
        fully instantiated. This is needed as we also check the devices of the weights, which are only available
        later after __init__. This allows to raise proper exceptions early before instantiating the full models
        if we know that the model does not support the requested attention.
aâ   does not support an attention implementation through torch.nn.functional.scaled_dot_product_attention yet. Please request the support for this architecture: https://github.com/huggingface/transformers/issues/28005. If you believe this error is a bug, please open an issue in Transformers GitHub repository and load your model with the argument `attn_implementation="eager"` meanwhile. Example: `model = AutoModel.from_pretrained("openai/whisper-tiny", attn_implementation="eager")`z2.4.1z¦Using the `SDPA` attention implementation on multi-gpu setup with ROCM may lead to performance issues due to the FA backend. Disabling it to use alternative backends.FT)Ú_supports_sdparÛ   rC  r²   r¯   r   Úhipr  Údevice_countÚparser   rÎ  rà  ÚbackendsÚenable_flash_sdp©rš   rp  s   &&r’   Ú_sdpa_can_dispatchÚ"PreTrainedModel._sdpa_can_dispatch  s±   € ð ×"×"Ð"ÜØ—>‘>×*Ñ*Ð+ð ,Oð Oóð ô �M‰M×ÑÒ)Ü—
‘
×'Ñ'Ó)¨AÔ-Ü—’œe×/Ñ/Ó0´7·=²=ÀÓ3IÔIä×Ñð yôô �N‰N×Ñ×0Ñ0°Ô7ár•   c                ó    <€ V ^8„  d   QhRS[ /# r‹   rŽ   )r�   r‘   s   "€r’   r“   rW  ;  s   ø€ ÷ 	ñ 	©$ñ 	r•   c                óv   € V P                  4       '       g#   \        V P                  P                   R24      hR# )z9
Check the availability of Grouped MM for a given model.
z1 does not support setting experts implementation.T)Ú_can_set_experts_implementationrÛ   rC  r²   r™   s   &r’   Ú_grouped_mm_can_dispatchÚ(PreTrainedModel._grouped_mm_can_dispatch;  s6   € ð
 ×3Ñ3×5Ò5Ü §¡× 7Ñ 7Ð8Ð8iÐjÓkÐkñ r•   c                ó&   <€ V ^8„  d   QhRS[ RS[ /# r  rŽ   )r�   r‘   s   "€r’   r“   rW  F  s   ø€ ÷ ñ ±Tð Ádñ r•   c                ó¤   € V P                   '       g#   \        V P                  P                   R24      h\	        4       '       g   \        R4      hR# )a  
Check the availability of Flex Attention for a given model.

Args:
    is_init_check (`bool`, *optional*):
        Whether this check is performed early, i.e. at __init__ time, or later when the model and its weights are
        fully instantiated. This is needed as we also check the devices of the weights, which are only available
        later after __init__. This allows to raise proper exceptions early before instantiating the full models
        if we know that the model does not support the requested attention.
aÄ   does not support an attention implementation through torch's flex_attention. Please request the support for this architecture: https://github.com/huggingface/transformers/issues/34809. If you believe this error is a bug, please open an issue in Transformers GitHub repository and load your model with the argument `attn_implementation="eager"` meanwhile. Example: `model = AutoModel.from_pretrained("openai/whisper-tiny", attn_implementation="eager")`z]PyTorch Flex Attention requirements in Transformers are not met. Please install torch>=2.5.0.T)Ú_supports_flex_attnrÛ   rC  r²   rg   r  r#  s   &&r’   Ú_flex_attn_can_dispatchÚ'PreTrainedModel._flex_attn_can_dispatchF  sY   € ð ×'×'Ð'ÜØ—>‘>×*Ñ*Ð+ð ,tð tóð ô ,×-Ò-ÜØoóð ñ
 r•   c                ó@   <€ V ^8„  d   QhRS[ R,          RS[RS[RS[ /# )rŒ   r×  Nrp  rq  r�   ©r­   r�   )r�   r‘   s   "€r’   r“   rW  a  s7   ø€ ÷ q.ñ q.Ù#&¨¥:ðq.Ù>Bðq.Ù_cðq.á	ñq.r•   c           	     ór  € \        V4      w  rET\        V RR4      ;'       g    . 9   d   RpVeh   \        V RR4      p\        VR7      '       dI   VeE   WV9  d?   V'       d   RV^ ,           2MV^ ,          p\        P	                  RV RV RV R	24       Tp\        V4      w  rETpR
p	\        VR7      '       dL   \
        P                  ! 4        F2  p
VRV
 28X  g   K  \
        V
,          R,          ! 4       '       d   K0  Rp	 M	  V P                  '       d\   V	'       dT   \        4       '       dD   \        4       '       g4   \        V,          p\        4       '       d
   VR8X  d   R
p	V'       d   RV 2p\        V4      '       dF    V'       d   \        WƒR7       M\        WƒR7       V	'       d   \        P	                  RV R24       V# V P%                  W‚4      p\        VR7      '       d   \        V4       V#   \         d4   pT	'       d%   \!        TR,          4      p
T P#                  Y¢R7       ThRp?ii ; i)a@  
Check that the `attn_implementation` exists and is supported by the models, and try to get the kernel from hub if
it matches hf kernels pattern.

Args:
    attn_implementation (`str` or `None`):
        The attention implementation to check for existence/validity.
    is_init_check (`bool`, *optional*):
        Whether this check is performed early, i.e. at __init__ time, or later when the model and its weights are
        fully instantiated. This is needed as we also check the devices of the weights, which are only available
        later after __init__. This allows to raise proper exceptions early before instantiating the full models
        if we know that the model does not support the requested attention.
    allow_all_kernels (`bool`, optional):
        Whether to load kernels from unverified hub repos, if `attn_implementation` is a custom kernel outside
        of the `kernels-community` hub repository.

Returns:
    `str`: The final attention implementation to use, including potential fallbacks from sdpa to eager, or from
    None to sdpa (to potentially eager).
Ú!_compatible_flash_implementationsNT©Ú"requested_attention_implementationzpaged|zNThis model is compatible with the following flash attention implementations: `z"`. Automatically falling back to `z` instead of `z`.FÚflash_attention_rö  Úflash_attention_2)rq  z/You do not have `flash_attn` installed, using `z%` from the `kernels` library instead!©rõ  rp  rM  )rm   r[  rl   rÎ  rà  rJ   r;  r  rf   rh   rK   ri   r=   rM   rL   rÍ  rÅ   r  Úget_correct_attn_implementation)rš   r×  rp  rq  Úis_pagedÚbase_implementationÚ compatible_flash_implementationsÚdefault_flash_implementationÚapplicable_attn_implementationÚrequested_original_flash_attnÚ
fa_versionr  s   &&&&        r’   rx  Ú5PreTrainedModel._check_and_adjust_attn_implementationa  sb  € ô. )GÐGZÓ([Ñ%ˆð ¤7¨4Ð1TÐVZÓ#[×#aÐ#aÐ_aÔbØ $Ðð Ò*Ü/6°tÐ=`ÐbfÓ/gÐ,ä,ÐPc×dÓdØ4Ò@Ø'ÔO÷ GO�fÐ=¸aÕ@ÐAÑBÐTtÐuvÕTwð -ô ×#Ñ#Ødð  fFð  eGð G6Ø6RÐ5SÐSaÐbuÐavÐvxðzôð 'CÐ#ä(FÐGZÓ([Ñ%ˆà)<Ð&à(-Ð%Ü'ÐK^×_Ó_äB×GÒGÖI�
ð (Ð-=¸j¸\Ð+JÖJÜ@ÀÕLÐMiÖj×lÔlà48Ð1Ùñ Jð ×%×%Ð%ß-Ü$×&Ò&Ü*×,Ò,ä-GÐH[Õ-\Ð*ä%×'Ò'Ð,?ÐCVÔ,Vð 16Ð-çØ39Ð:XÐ9YÐ1ZÐ.äÐ3×4Ò4ðçÜ5Ø6öô 0Ð0NÕt÷ 1Ü×'Ñ'ØIÐJhÐIið j>ð >ôð* .Ð-ð .2×-QÑ-QØ.ó.Ð*ô
 ,ÐOm×nÓnÜ+Ð,JÔKà-Ð-øô# ô ç0Ü!$Ð%8¸Õ%<Ó!=�JØ×1Ñ1ÀZÐ1Ômð �ûðús$   ÆG8 Æ G8 Æ-G8 Ç8H6È.H1È1H6c                ó4   <€ V ^8„  d   QhRS[ R,          RS[ /# )rŒ   rØ  Nr�   r­  )r�   r‘   s   "€r’   r“   rW  Ô  s!   ø€ ÷ 1ñ 1ÉsÐUYÍzð 1Ñ^añ 1r•   c                ó(   € V P                  V4      pV# )a  
Check that the `experts_implementation` exists and is supported by the models.

Args:
    experts_implementation (`str` or `None`):
        The experts implementation to check for existence/validity.
Returns:
    `str`: The final experts implementation to use.
)Ú"get_correct_experts_implementation)rš   rØ  Ú!applicable_experts_implementations   && r’   r|  Ú8PreTrainedModel._check_and_adjust_experts_implementationÔ  s   € ð -1×,SÑ,SÐTjÓ,kÐ)Ø0Ð0r•   c                ó:   <€ V ^8„  d   QhRS[ R,          RS[RS[ /# )rŒ   Úrequested_attentionNrp  r�   r1  )r�   r‘   s   "€r’   r“   rW  á  s(   ø€ ÷ $$ñ $$Á3ÈÅ:ð $$Ñ^bð $$Ñorñ $$r•   c                ó6  € Vf   RMTpVR.\         P                  4       ,           9  d®   RV R2pV P                  '       g   \        V RR4      '       d;   VR,          p\        P
                  ! 4        F  pVRV R	V R
2,          pK  	  VR R pV P                  '       d
   VR,          pV P                  '       d
   VR,          p\        VR,           4      h\        VR7      '       dN   \        P                  ! RV4      ;p'       d/   \        VP                  ^4      4      pV P                  WRR7       V# RV9   d   V P                  V4       V# RV9   d    V P!                  V4       V# V#   \        \"        3 d   pTe
   RT9   d   ThRp R p?T# R p?ii ; i)NÚsdpaÚeagerú Specified `attn_implementation="zc"` is not supported. The only possible arguments are `attn_implementation="eager"`, `"paged|eager"`Ú_supports_flash_attn_2Fú, z&`"attn_implementation=flash_attention_z0"`, `"attn_implementation=paged|flash_attention_z"`, zB, `"attn_implementation=sdpa"`, `"attn_implementation=paged|sdpa"`z(, `"attn_implementation=flex_attention"`rY  r4  z^flash_attention_(\d)$r8  Úflex_attentionéþÿÿÿ)ÚALL_ATTENTION_FUNCTIONSÚ
valid_keysr  r[  rJ   r;  r  r-  rÛ   rl   rƒ  r„  rÅ   Úgroupr  r.  r$  r  )rš   rH  rp  Úapplicable_attentionÚmessager@  Ú
fa_matchedr  s   &&&     r’   r9  Ú/PreTrainedModel.get_correct_attn_implementationá  sÒ  € Ø)<Ò)D™vÐJ]ÐØ¨ yÔ3J×3UÑ3UÓ3WÕ'WÔWà2Ð3GÐ2Hð IAð Að ð
 ×(×(Ð(¬G°DÐ:RÐTY×,ZÒ,ZØ˜4•�Ü"F×"KÒ"KÖ"M�JØÐ!GÈ
À|ð  TDð  EOð  DPð  PTð   Uõ  U’Gñ #Nà! # 2˜,�Ø×"×"Ð"ØÐ_Õ_�Ø×'×'Ð'ØÐEÕE�Ü˜W s�]Ó+Ð+ô (ÐK_×`Ó`ÜŸ)š)Ð$=Ð?SÓTÐTˆJÖTä˜Z×-Ñ-¨aÓ0Ó1ˆJØ×)Ñ)¸ZÐ)Ôeð $Ð#ð Ð!5Ô5Ø×(Ñ(¨Ô7ð $Ð#ð Ð+Ô+ð/Ø×'Ñ'¨Ô6ð $Ð#Ð#Ð#øô ¤Ð,ô /Ø&Ò2°vÐATÔ7TØ�GØ'.Ô$à#Ð#ûð/ús   ÅE- Å-FÅ>FÆFc                ó4   <€ V ^8„  d   QhRS[ R,          RS[ /# )rŒ   Úrequested_expertsNr�   r­  )r�   r‘   s   "€r’   r“   rW    s    ø€ ÷ "ñ "ÁCÈ$ÅJð "ÑSVñ "r•   c                óò  € Vf   RMTpR.\        \        \        P                  ! 4       4      \        \        P                  ! 4       4      ,          4      ,           pV Uu. uF	  pRV R2NK  	  ppRVR
,          ,           VR
&   RP                  V4      pW#9  d   RV RV R	2p\        V4      hVR8X  d    V P                  4        V# V# u upi   \        \        3 d   pTR8X  d   ThRp R p?T# R p?ii ; i)NÚ
grouped_mmrK  z`experts_implementation="z"`zand rN  z#Specified `experts_implementation="z5"` is not supported. The only possible arguments are rY  rM  )	r°   rf  r>   r;  r8   rË  rÛ   r)  r  )	rš   rY  Úapplicable_expertsÚbase_experts_fnsÚfnÚvalid_experts_str_listÚvalid_experts_strrU  r  s	   &&       r’   rD  Ú2PreTrainedModel.get_correct_experts_implementation  s  € Ø->Ò-F™\ÐL]ÐØ#˜9¤t¬CÔ0E×0JÒ0JÓ0LÓ,MÔPSÔTm×TrÒTrÓTtÓPuÕ,uÓ'vÕvÐÙO_Ó!`ÑO_ÈÐ$=¸b¸TÀÓ"DÑO_ÐÐ!`Ø%+Ð.DÀRÕ.HÕ%HÐ˜rÑ"Ø ŸI™IÐ&<Ó=ÐØÔ5à5Ð6HÐ5IÐI~Ø$Ð% Qð(ð ô ˜WÓ%Ð%ð  Ô-ð-Ø×-Ñ-Ô/ð "Ð!Ð!Ð!ùò' "aøô ¤Ð,ô -Ø$¨Ô4Ø�GØ%,Ô"à!Ð!ûð-ús   ÁC
Â6C ÃC6Ã C1Ã1C6c                ó    <€ V ^8„  d   QhRS[ /# r‹   rŽ   )r�   r‘   s   "€r’   r“   rW     s   ø€ ÷ ñ ©Tñ r•   c                ó‚  € \         P                  P                  V P                  4      pVe   \	        VR4      '       g   R# VP
                  p\        VRRR7      ;_uu_ 4       pVP                  4       pRRR4       \        P                  ! RX4      '       d   RV9   ;'       d    R	V9   # R
#   + '       g   i     LA; i)zµDetect whether the class supports setting its attention implementation dynamically. It is an ugly check based on
opening the file, but avoids maintaining yet another property flag.
NÚ__file__FÚrr  r  zclass \w+Attention\(nn.Module\)Úeager_attention_forwardz&ALL_ATTENTION_FUNCTIONS.get_interface(T)
r  ÚmodulesrÍ   r³   rÀ   rd  r   r8  rƒ  r„  ©rg  Úclass_moduleÚ
class_filerF  Úcodes   &    r’   Ú_can_set_attn_implementationÚ,PreTrainedModel._can_set_attn_implementation  s—   € ô
 —{‘{—‘ s§~¡~Ó6ˆàÒ¤w¨|¸Z×'HÒ'HÙØ!×*Ñ*ˆ
Ü�*˜c¨G×4Õ4¸Ø—6‘6“8ˆD÷ 5ô �9Š9Ð7¸×>Ò>Ø,°Ñ4×iÐiÐ9aÐeiÑ9iÐiñ ÷ 5×4ús   Á$B.Â.B>	c                ó    <€ V ^8„  d   QhRS[ /# r‹   rŽ   )r�   r‘   s   "€r’   r“   rW  3  s   ø€ ÷ 5ñ 5±ñ 5r•   c                ó2  € \         P                  P                  V P                  4      pVe   \	        VR4      '       g   R# VP
                  p\        VRRR7      ;_uu_ 4       pVP                  4       pRRR4       RV9   #   + '       g   i     RX9   # ; i)z³Detect whether the class supports setting its experts implementation dynamically. It is an ugly check based on
opening the file, but avoids maintaining yet another property flag.
Nrd  Fre  r  r  z@use_experts_implementation)r  rg  rÍ   r³   rÀ   rd  r   r8  rh  s   &    r’   r(  Ú/PreTrainedModel._can_set_experts_implementation2  s~   € ô
 —{‘{—‘ s§~¡~Ó6ˆàÒ¤w¨|¸Z×'HÒ'HÙØ!×*Ñ*ˆ
Ü�*˜c¨G×4Õ4¸Ø—6‘6“8ˆD÷ 5ð -°Ñ4Ð4÷ 5Ö4ð -°Ñ4Ð4ús   Á$BÂB	c                ó6   <€ V ^8„  d   QhRS[ S[,          RS[/# )rŒ   r×  rq  )r­   r®   r�   )r�   r‘   s   "€r’   r“   rW  A  s$   ø€ ÷ d8ñ d8¹3Á½:ð d8ÑZ^ñ d8r•   c                ó®  € \        V\        4      '       g   TM%VP                  RV P                  P                  4      pW0P                  P                  8w  dh   V P                  4       '       g.   \        P                  V P                  P                   R24       M$V P                  VRVR7      pW0P                  n        V P                  4        EFg  pW@Jg   K  \        V\        4      '       g   K#  VP                  P                  V P                  P                  8w  g   KT  \        VP                  R4      '       d   Kr  VP                  4       '       g.   \        P                  VP                  P                   R24       M¡Tp\        V\        4      '       di   V P                  P                   FN  p\!        V P                  V4      VP                  J g   K)  VP                  WdP                  P                  4      p M	  VP#                  V4      pWTP                  n        RVP                  n        EKj  	  V P                  P                   Fÿ  p\!        V P                  V4      ;pf   K  \        V\        4      '       g   TMVP                  WgP                  4      p\        VR4      '       g…   WWP                  8w  du   VR.\&        P)                  4       ,           9  d0   \+        R	V R
V R\-        \&        P)                  4       4       24      hWWn        \        P                  RV RV R24       Kè  \        VR4      '       g   Kü  V=EK  	  R# )aS  
Set the requested `attn_implementation` for this model.

Args:
    attn_implementation (`str` or `dict`):
        The attention implementation to set for this model. It can be either a `str`, in which case it will be
        dispatched to all submodels if relevant, or a `dict` where keys are the sub_configs name, in which case each
        submodel will dispatch the corresponding value.
    allow_all_kernels (`bool`, optional):
        Whether to load kernels from unverified hub repos, if `attn_implementation` is a custom kernel outside
        of the `kernels-community` hub repository.
rÀ  zØ does not support setting its attention implementation dynamically, because it does not follow the functional approach based on AttentionInterface (see https://huggingface.co/docs/transformers/en/attention_interface)Fro  Ú_attn_was_changedTNrK  rL  z"` is not supported for zd. The only possible arguments are "eager" (manual attention implementation)or one of the following: z8We set the attention implementation for the sub-config `z` to `zŽ` without finding the associated sub-model. For this reason we could not check if the model supports it. You may encounter undefined behavior.)r‡  r®   rÍ   rÜ  ry  rl  rÎ  rñ  rC  r²   rx  r{  rg  r„   rÀ   rá  r[  r9  rs  rQ  rR  rÛ   r°   )rš   r×  rq  Úrequested_implementationr_  Úsub_implementationÚsubconfig_keyÚ	subconfigs   &&&     r’   Úset_attn_implementationÚ'PreTrainedModel.set_attn_implementationA  sö  € ô Ð1´4×8Ò8ñ  à$×(Ñ(¨¨T¯[©[×-MÑ-MÓNð 	!ð $§{¡{×'GÑ'GÔGà×4Ñ4×6Ò6Ü—‘Ø—~‘~×.Ñ.Ð/ð 0\ð \õð ,0×+UÑ+UØ,¸EÐUfð ,Vó ,Ð(ð =U—‘Ô9ð Ÿ™ŸˆIð Õ%Ü˜y¬/×:Ô:Ø×$Ñ$×.Ñ.°$·+±+×2GÑ2GÖGä 	× 0Ñ 0Ð2E×FÔFð !×=Ñ=×?Ò?Ü—N‘NØ$×.Ñ.×7Ñ7Ð8ð 9`ð `õð *BÐ&Ü!Ð"5´t×<Ò<Ø-1¯[©[×-DÔ-D˜Mä& t§{¡{°MÓBÀi×FVÑFVÕVØ5H×5LÑ5LØ$1×3CÑ3C×3XÑ3Xó6"Ð 2ñ !&ñ .Eð *3×)RÑ)RÐSeÓ)fÐ&ØEW×$Ñ$ÔBð 6:�	× Ñ ×2ñE (ðJ "Ÿ[™[×4Ô4ˆMÜ$ T§[¡[°-Ó@Ð@�	ÔMô &Ð&9¼4×@Ò@ñ -à,×0Ñ0°×@^Ñ@^Ó_ð #ô   	Ð+>×?Ò?à*×.LÑ.LÔLà)°'°Ô=T×=_Ñ=_Ó=aÕ1aÔaÜ(Ø>Ð?QÐ>RÐRjÐkxÐjyð z8ä8<Ô=T×=_Ñ=_Ó=aÓ8bÐ7cðeóð ð
 ?QÔ;Ü—N‘NØRÐS`ÐRaÐagÐhzÐg{ð |@ð @öô ˜yÐ*=×>Ô>Ø%Ó7ó9 5r•   c                ó0   <€ V ^8„  d   QhRS[ S[,          /# )rŒ   rØ  )r­   r®   )r�   r‘   s   "€r’   r“   rW  §  s   ø€ ÷ 6Wñ 6WÁÁtÅñ 6Wr•   c                óx  € \        V\        4      '       g   TM%VP                  RV P                  P                  4      pV P                  P                  pRW239   d   W28w  d   \        RV: RV: R24      hW P                  P                  8w  d"   V P                  V4      pW P                  n        V P                  4        Fô  pW@Jg   K
  \        V\        4      '       g   K"  VP                  P                  V P                  P                  8w  g   KS  Tp\        V\        4      '       di   V P                  P                   FN  p\        V P                  V4      VP                  J g   K)  VP                  WdP                  P                  4      p M	  VP                  V4      pWTP                  n        Kö  	  R# )a‹  
Set the requested `experts_implementation` for this model.

Args:
    experts_implementation (`str` or `dict`):
        The experts implementation to set for this model. It can be either a `str`, in which case it will be
        dispatched to all submodels if relevant, or a `dict` where keys are the sub_configs name, in which case each
        submodel will dispatch the corresponding value.
rÀ  Údeepgemm_megamoez*Cannot switch experts implementation from ú to z at runtime: `deepgemm_megamoe` is a load-time choice. Reload via `from_pretrained(..., experts_implementation=...)` to switch.N)r‡  r®   rÍ   rÜ  r}  r�  r|  r~  rg  r„   rC  rá  r[  rD  )rš   rØ  rt  Úcurrentr_  ru  rv  s   &&     r’   Úset_experts_implementationÚ*PreTrainedModel.set_experts_implementation§  s|  € ô Ð4´d×;Ò;ñ #à'×+Ñ+¨B°·±×0SÑ0SÓTð 	!ð —+‘+×5Ñ5ˆØ 'Ð!DÔDÈÔIlÜØ<¸W¹KÀtÐLdÑKgð hPð Póð ð $§{¡{×'JÑ'JÔJØ'+×'TÑ'TÐUmÓ'nÐ$à;S�K‰KÔ8ð Ÿ™žˆIð Õ%Ü˜y¬/×:Ô:Ø×$Ñ$×.Ñ.°$·+±+×2GÑ2GÖGð &>Ð"ÜÐ4´d×;Ò;Ø)-¯©×)@Ô)@˜ä" 4§;¡;°Ó>À)×BRÑBRÕRØ1G×1KÑ1KØ -×/?Ñ/?×/WÑ/Wó2Ð.ñ "ñ *Að &/×%QÑ%QÐRdÓ%eÐ"ØDV× Ñ ÖAó) (r•   c                óR  € R p. p\        4       pRpV P                  4        Fœ  p\        V\        4      '       d   \	        VR4      '       g   K-   VP                  4       pTe   \	        TR4      '       g   KV  \        T4      pYs9   d   Ki  TP                  T4       TP                  TP                  T4      4       RpKž  	  W n        V'       d   V^ ,          V n        V'       g/   \        P                  V P                  P                    R24       R# R#   \         d     EK  i ; i)z‡
Enables the gradients for the input embeddings. This is useful for fine-tuning adapter weights while keeping
the model weights fixed.
c                 ó(   € VP                  R 4       R# )TN)Úrequires_grad_)rU  ÚinputÚoutputs   &&&r’   Úmake_inputs_require_gradsÚMPreTrainedModel.enable_input_require_grads.<locals>.make_inputs_require_gradså  s   € Ø×!Ñ! $Ö'r•   Fr?  NÚregister_forward_hookTa   does not expose input embeddings. Gradients cannot flow back to the token embeddings when using adapters or gradient checkpointing. Override `get_input_embeddings` to fully support those features, or set `_input_embed_layer` to the attribute name that holds the embeddings.)rf  rg  r‡  r„   rÀ   r?  rB  rˆ  rk  ri  rˆ  Ú_require_grads_hooksÚ_require_grads_hookrÎ  rà  rC  r²   )rš   r†  ÚhooksÚseen_modulesÚfound_embeddingsrU  Úinput_embeddingsÚembedding_ids   &       r’   Úenable_input_require_gradsÚ*PreTrainedModel.enable_input_require_gradsß  s  € ò	(ð ˆÜ“uˆØ Ðà—l‘l–nˆFÜ˜v¤×7Ò7¼GÀFÐLb×<cÒ<cÙðØ#)×#>Ñ#>Ó#@Ð ð  Ò'¬wÐ7GÐI`×/aÒ/aÙäÐ.Ó/ˆLØÔ+Ùà×Ñ˜\Ô*Ø�L‰LÐ)×?Ñ?Ð@YÓZÔ[Ø#Òñ% %ð( %*Ô!ßà',¨Q¥xˆDÔ$ßÜ×ÑØ—>‘>×*Ñ*Ð+ð ,wð wöñ  øô% 'ô Ûðús   ÁDÄD&Ä%D&c                ó    € \        V RR4      pV'       g   R# V F  pVP                  4        K  	  . V n        \        V R4      '       d   V =R# R# )z$
Removes the `_require_grads_hook`.
r‰  NrŠ  )r[  Úremover‰  rÀ   rŠ  )rš   r‹  Úhooks   &  r’   Údisable_input_require_gradsÚ+PreTrainedModel.disable_input_require_grads	  sO   € ô ˜Ð4°dÓ;ˆßÙãˆDØ�K‰KŽMñ ð %'ˆÔ!Ü�4Ð.×/Ò/ØÒ(ñ 0r•   c                ó.   <€ V ^8„  d   QhRS[ R,          /# ©rŒ   ÚmodalityNr­  )r�   r‘   s   "€r’   r“   rW  	  s   ø€ ÷ ñ ¡C¨$¥Jñ r•   c                óf  € VR9   d   . ROpM#VR8X  d   . R	OpMVf   RR.pM\        RV 24      hV F!  p\        W4      '       g   K  \        W4      u # 	  V P                  V JdK   \        V P                  R4      '       d/   V P                  P	                  VR7      pW@P                  8w  d   V# V # )
aA  
Best-effort lookup of the *encoder* module. If provided with `modality` argument,
it looks for a modality-specific encoder in multimodal models (e.g. "image_encoder")
By default the function returns model's text encoder if any, and otherwise returns `self`.

Possible `modality` values are "image", "video" and "audio".
ÚaudioÚtext_encoderÚencoderúHUnnrecognized modality, has to be "image", "video" or "audio" but found Úget_encoder©r™  ©ÚimageÚvideo©Úvision_towerÚvisualÚvision_modelÚvision_encoderÚimage_tower)Úaudio_towerÚaudio_encoderÚspeech_encoder)rÛ   rÀ   r[  r@  rŸ  )rš   r™  Úpossible_module_namesr^  Úbase_encoders   &&   r’   rŸ  ÚPreTrainedModel.get_encoder	  sµ   € ð Ð)Ô)Ú$oÑ!Ø˜Ô Ú$VÑ!ØÒØ%3°YÐ$?Ñ!äÐgÐhpÐgqÐrÓsÐsã)ˆDÜ�t×"Ô"Ü˜tÓ*Ò*ñ *ð �?‰? $Ó&¬7°4·?±?ÀM×+RÒ+RØŸ?™?×6Ñ6ÀÐ6ÓIˆLð Ÿ™Ô.Ø#Ð#ð ˆr•   c                ó.   <€ V ^8„  d   QhRS[ R,          /# r˜  r­  )r�   r‘   s   "€r’   r“   rW  :	  s   ø€ ÷ %ñ %©S°4­Zñ %r•   c                óZ  € VR
9   d   . ROpM#VR8X  d   RR.pMVf   RR.pM\        RV 24      hV F#  p\        W4      '       g   K  \        WV4        R# 	  V P                  V JdC   \        V P                  R4      '       d   V P                  P	                  WR	7       R# Wn        R# R# )zC
Symmetric setter. Mirrors the lookup logic used in `get_encoder`.
r›  rª  r«  Nrœ  r�  rž  Úset_encoderr   r¡  r¤  )rÛ   rÀ   r¦  r@  r²  r~  )rš   r�  r™  r­  r^  s   &&&  r’   r²  ÚPreTrainedModel.set_encoder:	  s¬   € ð Ð)Ô)Ú$oÑ!Ø˜Ô Ø%2°OÐ$DÑ!ØÒØ%3°YÐ$?Ñ!äÐgÐhpÐgqÐrÓsÐsã)ˆDÜ�t×"Ô"Ü˜ GÔ,Úñ *ð
 �?‰? $Ó&Ü�t—‘¨×6Ò6Ø—‘×+Ñ+¨GÐ+ÖGà$–
ñ	 'r•   c                óè   € . ROpV F!  p\        W4      '       g   K  \        W4      u # 	  V P                  V Jd7   \        V P                  R4      '       d   V P                  P                  4       # V # )ad  
Best-effort lookup of the *decoder* module.

Order of attempts (covers ~85 % of current usages):

1. `self.decoder/self.language_model/self.text_model`
2. `self.base_model`                  (many wrappers store the decoder here)
3. `self.base_model.get_decoder()`    (nested wrappers)
4. fallback: raise for the few exotic models that need a bespoke rule
Úget_decoder)r>  Ú
text_modelÚdecoderÚtext_decoder)rÀ   r[  r@  rµ  )rš   r­  r^  s   &  r’   rµ  ÚPreTrainedModel.get_decoderT	  sc   € ò !\ÐÛ)ˆDÜ�t×"Ô"Ü˜tÓ*Ò*ñ *ð �?‰? $Ó&¬7°4·?±?ÀM×+RÒ+RØ—?‘?×.Ñ.Ó0Ð0ð ˆr•   c                ó  € . ROpV F#  p\        W4      '       g   K  \        WV4        R# 	  V P                  V JdB   \        V P                  R4      '       d   V P                  P                  V4       R# Wn        R# R# )zC
Symmetric setter. Mirrors the lookup logic used in `get_decoder`.
NÚset_decoder)r>  r¶  r·  )rÀ   r¦  r@  r»  r~  )rš   r·  r­  r^  s   &&  r’   r»  ÚPreTrainedModel.set_decoderk	  sh   € ò
 !LÐÛ)ˆDÜ�t×"Ô"Ü˜ GÔ,Úñ *ð
 �?‰? $Ó&Ü�t—‘¨×6Ò6Ø—‘×+Ñ+¨GÖ4à$–
ñ	 'r•   c           	     óø	  € \        V P                  R4      '       d"   V P                  P                  ;'       g    RpM‹\        V P                  R4      '       d   V P                  P                  pMX\        V P                  R4      '       d   V P                  P                  pM%\        V P                  P                  4       RR4      p\        V\        P                  \        P                  \        P                  \        P                  \        P                  \        P                  34      '       de   \        VRR4      e$   \        P                   ! VP"                  RVR7       VP$                  e#   \        P&                  ! VP$                  4       R# R# \        V\        P(                  4      '       d[   VP+                  4        FD  w  r4RV9   d   \        P,                  ! V4       K$  R	V9   g   K-  \        P.                  ! VR4       KF  	  R# \        V\        P0                  4      '       d†   \        P                   ! VP"                  RVR7       VP2                  eS   \        VP"                  R
R4      '       g4   \        P&                  ! VP"                  VP2                  ,          4       R# R# R# \        V\        P4                  4      '       d   VP7                  4        R# \        V\        P8                  \        P:                  \        P<                  \        P>                  34      '       g7   RVP@                  PB                  9   g   RVP@                  PB                  9   dÒ   \        VRR4      e!   \        PD                  ! VP"                  4       \        VR	R4      e!   \        P&                  ! VP$                  4       \        VRR4      ec   \        P&                  ! VPF                  4       \        PD                  ! VPH                  4       \        P&                  ! VPJ                  4       R# R# RVP@                  PB                  9   d¡   \        VR4      '       d�   VPL                  R8w  d   \N        VPL                  ,          MVPP                  pV! VP                  4      w  rg\        PR                  ! VPT                  V4       \        PR                  ! VPV                  V4       R# R# R# )aD  
Initialize the weights. This is quite general on purpose, in the spirit of what we usually do. For more complex
initialization scheme, it should be overridden by the derived `PreTrainedModel` class. In case a model adds an explicit
`nn.Parameter`, this method should also be overridden in order to initialize it correctly.
Úinitializer_rangeg{®Gáz”?Úinit_stdÚinitializer_factorÚweightNg        ©ÚmeanÚstdÚbiasÚ_is_hf_initializedFÚ	LayerNormÚRMSNormÚrunning_meanÚRotaryEmbeddingÚoriginal_inv_freqÚdefault),rÀ   rÜ  r¾  r¿  rÀ  r[  Úget_text_configr‡  r   ÚLinearÚConv1dÚConv2dÚConv3dÚConvTranspose1dÚConvTranspose2drÞ  Únormal_rÁ  rÅ  Úzeros_ÚLSTMr(  Úxavier_uniform_Ú	constant_r&  Úpadding_idxÚMultiheadAttentionÚ_reset_parametersÚ	GroupNormÚBatchNorm1dÚBatchNorm2dÚBatchNorm3drC  r²   Úones_rÉ  Úrunning_varÚnum_batches_trackedÚ	rope_typerN   Úcompute_default_rope_parametersÚcopy_Úinv_freqrË  )rš   rU  rÄ  r^  rï  Úrope_fnÚbuffer_valuerp  s   &&      r’   Ú_init_weightsÚPreTrainedModel._init_weights|	  sN  € ô �4—;‘;Ð 3×4Ò4Ø—+‘+×/Ñ/×7Ð7°4‰CÜ�T—[‘[ *×-Ò-Ø—+‘+×&Ñ&‰CÜ�T—[‘[Ð"6×7Ò7Ø—+‘+×0Ñ0‰Cô ˜$Ÿ+™+×5Ñ5Ó7Ð9LÈdÓSˆCä�fœrŸy™y¬"¯)©)´R·Y±YÄÇ	Á	Ì2×K]ÑK]Ô_a×_qÑ_qÐr×sÒsÜ�v˜x¨Ó.Ò:Ü—’˜VŸ]™]°¸#Õ>Ø�{‰{Ò&Ü—’˜FŸK™KÖ(ñ 'ä˜¤§¡×(Ò(Ø%×6Ñ6Ö8‘�Ø˜tÔ#Ü×(Ò(¨Ö/Ø˜t–^Ü—N’N 5¨#Ö.ó	  9ô
 ˜¤§¡×-Ò-Ü�LŠL˜Ÿ™¨S°cÕ:à×!Ñ!Ò-´g¸f¿m¹mÐMaÐch×6iÒ6iÜ—’˜FŸM™M¨&×*<Ñ*<Õ=Ö>ñ 7jÑ-ä˜¤× 5Ñ 5×6Ò6à×$Ñ$Ö&ô �v¤§¡¬b¯n©n¼b¿n¹nÌbÏnÉnÐ]×^Ò^Ø˜f×.Ñ.×7Ñ7Ô7Ø˜F×,Ñ,×5Ñ5Ô5ô �v˜x¨Ó.Ò:Ü—
’
˜6Ÿ=™=Ô)Ü�v˜v tÓ,Ò8Ü—’˜FŸK™KÔ(ä�v˜~¨tÓ4Ò@Ü—’˜F×/Ñ/Ô0Ü—
’
˜6×-Ñ-Ô.Ü—’˜F×6Ñ6Ö7ñ Að
  &×"2Ñ"2×";Ñ";Ô;ÄÈÐPc×@dÒ@dð ×#Ñ# yÔ0ô $ F×$4Ñ$4Ö5à×;Ñ;ð ñ
 & f§m¡mÓ4‰OˆLÜ�JŠJ�v—‘¨Ô5Ü�JŠJ�v×/Ñ/°Ö>ñ AeÑ;r•   c                ó    <€ V ^8„  d   QhRS[ /# )rŒ   r´  rŽ   )r�   r‘   s   "€r’   r“   rW  ¼	  s   ø€ ÷ )ñ )¹$ñ )r•   c                óê  € \        VRR4      '       d   R# V'       d¾   \        ;QJ d,    R VP                  RR7       4       F  '       d   K   RM!	  RM! R VP                  RR7       4       4      '       dd   \        ;QJ d,    R VP                  RR7       4       F  '       d   K   RM!	  RM! R VP                  RR7       4       4      '       d
   RVn        R# V P                  V4       RVn        R# )z=
Initialize the weights if they are not already initialized.
rÆ  FNc              3   ó<   "  € T F  p\        VR R4      x € K  	  R# 5i)rÆ  FN©r[  rî  s   & r’   r  Ú6PreTrainedModel._initialize_weights.<locals>.<genexpr>È	  s   é € ÐnÑMmÀE”G˜EÐ#7¸×?Ð?ÓMmùs   ‚)ÚrecurseTc              3   óH   "  € T F  pVf   K	  \        VRR4      x € K  	  R # 5i)NrÆ  Frî  )r  Úbuffers   & r’   r  rï  É	  s)   é € ð á;�FØô =”˜Ð 4°e×<Ð<Û;ùs   ‚"�")r[  Úallrò  ÚbuffersrÆ  ré  )rš   rU  r´  s   &&&r’   Ú_initialize_weightsÚ#PreTrainedModel._initialize_weights¼	  sÃ   € ô �6Ð/°×7Ò7Ù÷ ß“ÑnÈV×M^ÑM^ÐglÐM^ÔMmÓn——’ÑnÈV×M^ÑM^ÐglÐM^ÔMmÓn×nÒnß“ñ à$Ÿn™n°U˜nÔ;ó——’ñ à$Ÿn™n°U˜nÔ;ó÷ ò ð )-ˆFÔ%Ùà×Ñ˜6Ô"Ø$(ˆÖ!r•   c                ó  a€ \        \        P                  P                  R4      '       g/   R V3R llo\	        \        P                  P                  RS4       \        V R4      pV! V P                  V P                  4       4       R# )aå  
This is equivalent to calling `self.apply(self._initialize_weights)`, but correctly handles composite models.
This function dynamically dispatches the correct `init_weights` function to the modules as we advance in the
module graph along the recursion. It can handle an arbitrary number of sub-models. Without it, every composite
model would have to recurse a second time on all sub-models explicitly in the outer-most `_init_weights`, which
is extremely error prone and inefficient.
Úsmart_applyc                óŠ   € V ^8„  d   QhR\         P                  R\        \         P                  \        .R3,          R\        /# )rŒ   rU  r^  Nr´  )r   rV  r   r�   )r�   s   "r’   r“   Ú8PreTrainedModel.initialize_weights.<locals>.__annotate__â	  s9   € ÷ ñ ¤B§I¡Ið ´8¼R¿Y¹YÌÐ<MÈtÐ<SÕ3Tð Ôfjñ r•   c                 ó®   <€ V P                  4        F7  p\        V\        4      '       d   S! W3P                  V4       K.  S! W1V4       K9  	  V! W4       V # r—   )Úchildrenr‡  r„   rõ  )rU  r^  r´  Úchildrø  s   &&& €r’   rø  Ú7PreTrainedModel.initialize_weights.<locals>.smart_applyâ	  sJ   ø€ Ø#Ÿ_™_Ö.�Eä! %¬×9Ò9Ù# E×+DÑ+DÀnÖUá# E¨~Ö>ñ /ñ �6Ô*Ø�r•   N)rÀ   r¯   r   rV  r¦  r[  rõ  r´  )rš   Úsmart_apply_fnrø  s   & @r’   Úinitialize_weightsÚ"PreTrainedModel.initialize_weightsÕ	  sa   ø€ ô ”u—x‘x—‘¨×6Ò6÷ð ô ”E—H‘H—O‘O ]°KÔ@ô !  }Ó5ˆá�t×/Ñ/°×1DÑ1DÓ1FÖGr•   c                ó&   <€ V ^8„  d   QhRS[ RS[/# )rŒ   rŠ  r�   )r�   r®   )r�   r‘   s   "€r’   r“   rW  ó	  s   ø€ ÷ p%ñ p%¹Dð p%ÉTñ p%r•   c                ó°  aaa€ V'       d�   / pV P                  RR7       Fu  w  r4\        V\        4      '       g   K  VP                  RR7      pVR8w  d/   VP	                  4        UUu/ uF  w  rgV RV 2V RV 2bK  	  pppVP                  V4       Kw  	  V# V P                  p\        V P                  RR4      p	V	'       g   / # Vf   / # \        P                  ! R4      o\        ;QJ dB    V3R lVP                  4       VP                  4       ,           4       F  '       d   K   RM7	  R	M3! V3R lVP                  4       VP                  4       ,           4       4      '       d   VP                  4       # / pV P                  RR7       UU
u0 uF  w  rjVkK	  	  up
pV P!                  RR7       UU
u0 uF  w  rjVkK	  	  up
p,          pVP	                  4        Fã  w  ooR
S,           oR
S,           o\#        \%        V3R lV4      4      p\#        \%        V3R lV4      4      p\'        V4      ^ 8”  d1   \'        V4      ^ 8”  d!   \'        V4      \'        V4      ,          ^ 8w  d   \)        RS RS RV RV 24      h\+        V\-        V4      4       F)  w  rïWòP                  4       9   d   W/,          W.&   K%  WòV&   K+  	  Kå  	  V# u uppi u up
pi u up
pi )a‡	  
Return the expanded tied weight keys (in case they contain modules or regex patterns) for only the current
model, or recursively for all submodels if `all_submodels=True` (i.e. it will re-check the config values for all
submodels).

For almost all models, we only require to tie the embeddings, so the model has an internal property
`_tied_weights_keys = {"lm_head.weight": "model.embed_tokens.weight"}`. In this case, the mapping is already
"expanded", i.e. it already contains full parameters, and this function will simply return a copy of the property.
For more complex patterns, e.g. for `DFineForObjectDetection`, we have the following attribute
```
_tied_weights_keys = {
    r"bbox_embed.(?![0])\d+": "bbox_embed.0",
    r"class_embed.(?![0])\d+": "class_embed.0",
    "model.decoder.class_embed": "class_embed",
    "model.decoder.bbox_embed": "bbox_embed",
}
```
In this case, the function looks up all the model's parameters and buffers, and matches all the params,
returning the following:
```
{
    'bbox_embed.1.layers.0.bias': 'bbox_embed.0.layers.0.bias',
    'bbox_embed.1.layers.0.weight': 'bbox_embed.0.layers.0.weight',
    'bbox_embed.1.layers.1.bias': 'bbox_embed.0.layers.1.bias',
    'bbox_embed.1.layers.1.weight': 'bbox_embed.0.layers.1.weight',
    'bbox_embed.1.layers.2.bias': 'bbox_embed.0.layers.2.bias',
    'bbox_embed.1.layers.2.weight': 'bbox_embed.0.layers.2.weight',
    'bbox_embed.2.layers.0.bias': 'bbox_embed.0.layers.0.bias',
    'bbox_embed.2.layers.0.weight': 'bbox_embed.0.layers.0.weight',
    ...
    'class_embed.1.bias': 'class_embed.0.bias',
    'class_embed.1.weight': 'class_embed.0.weight',
    'class_embed.2.bias': 'class_embed.0.bias',
    'class_embed.2.weight': 'class_embed.0.weight',
    ...
    'model.decoder.class_embed.0.bias': 'class_embed.0.bias',
    'model.decoder.class_embed.0.weight': 'class_embed.0.weight',
    'model.decoder.class_embed.1.bias': 'class_embed.0.bias',
    'model.decoder.class_embed.1.weight': 'class_embed.0.weight',
    ...
    'model.decoder.bbox_embed.0.layers.0.bias': 'bbox_embed.0.layers.0.bias',
    'model.decoder.bbox_embed.0.layers.0.weight': 'bbox_embed.0.layers.0.weight',
    'model.decoder.bbox_embed.0.layers.1.bias': 'bbox_embed.0.layers.1.bias',
    'model.decoder.bbox_embed.0.layers.1.weight': 'bbox_embed.0.layers.1.weight',
    ...
}
```
i.e. all the parameters matching the regex and modules patterns in `_tied_weights_keys`
F)Úremove_duplicater‰  rÀ  rY  Útie_word_embeddingsz ^[A-Za-z0-9_\.]+(weight)|(bias)$c              3   óF   <"  € T F  pSP                  V4      x € K  	  R # 5ir—   )r»  )r  rD  Úcommon_case_regexs   & €r’   r  ÚAPreTrainedModel.get_expanded_tied_weights_keys.<locals>.<genexpr>@
  s"   øé € Ð_Ñ3^¨aÐ ×&Ñ& q×)Ð)Ó3^ùs   ƒ!TÚ^c                 ó2   <€ \         P                  ! SV 4      # r—   r‚  )ÚxÚsource_names   &€r’   r  Ú@PreTrainedModel.get_expanded_tied_weights_keys.<locals>.<lambda>L
  ó   ø€ ´B·I²I¸kÈ1Ô4Mr•   c                 ó2   <€ \         P                  ! SV 4      # r—   r‚  )r  Útarget_names   &€r’   r  r  M
  r  r•   zAThere is an issue with your definition of `tie_weights_keys` for Ú:z. We found z to tie into )rZ  r‡  r„   rš  r9  rœ  rX  r[  rÜ  rƒ  Úcompileró  r;  rì   r—  r(  Únamed_buffersr!  Úfilterrí   rÛ   r  r   )rš   rŠ  Úexpanded_tied_weightsÚprefixr_  Úsubmodel_tied_weightsrD  rE  Útied_mappingr  rp  Úall_param_namesÚsource_paramsÚtarget_paramsÚtarget_nÚsource_nr  r  r  s   &&              @@@r’   rš  Ú.PreTrainedModel.get_expanded_tied_weights_keysó	  sÆ  ú€ ÷d Ø$&Ð!Ø%)×%7Ñ%7ÈÐ%7Ö%OÑ!�Ü˜i¬×9Ô9à,5×,TÑ,TÐchÐ,TÓ,iÐ)Ø ”|àI^×IdÑIdÔIfô1ÙIfÁÀ˜v˜h a¨ s˜O°¨x°q¸¸¨_Ò<ÑIfð .ñ 1ð *×0Ñ0Ð1FÖGñ &Pð )Ð(à×.Ñ.ˆô & d§k¡kÐ3HÈ%ÓPÐß"ØˆIàÒ!ØˆIô ŸJšJÐ'JÓKÐß‹3Ô_°<×3DÑ3DÓ3FÈ×I\ÑI\ÓI^Ö3^Ó_�3�3Š3Ô_°<×3DÑ3DÓ3FÈ×I\ÑI\ÓI^Ö3^Ó_×_Ò_Ø×$Ñ$Ó&Ð&ð !#ÐØ)-×)>Ñ)>ÐPUÐ)>Ô)VÔWÑ)V¡ ›1Ñ)VÒWØ×,Ñ,¸eÐ,ÔDô[
ÙD‘$�!‹AÑDò[
õ 
ˆð )5×(:Ñ(:Ö(<Ñ$ˆK˜Ø Õ+ˆKØ Õ+ˆKä"¤6Ô*MÈÓ#_Ó`ˆMÜ"¤6Ô*MÈÓ#_Ó`ˆMä˜Ó&¨Ô*Ü˜=Ó)¨AÔ-Ü�}Ó%¬¨MÓ(:Õ:¸aÔ?ä ØWÐXcÐWdÐdeÐfqÐerð s Ø -˜¨m¸M¸?ðLóð ô
 '*¨-¼¸}Ó9MÖ&NÑ"�ð ×9Ñ9Ó;Ô;à6KÕ6UÐ)Ó3ð 7?¨(Ó3ó 'Oñ! )=ð6 %Ð$ùóo1ùó2 Xùó [
s   Á)KÆKÆ6KTc                óD   <€ V ^8„  d   QhRS[ S[,          R,          RS[/# )rŒ   Úmissing_keysNÚrecompute_mapping)rf  r­   r�   )r�   r‘   s   "€r’   r“   rW  e
  s(   ø€ ÷ Q8ñ Q8©©C­°4­ð Q8ÑSWñ Q8r•   c                ó¾  € V'       g   V P                   pMV P                  RR7      p\        VP                  4       4      p\	        V4       EFŒ  w  pw  rVVeý   RpWa9  pWQ9  p	V'       d~   V	'       dv   \
        P                  ! V P                  V4      V P                  V4      4      '       g:   \        P                  RV RV R24       V P                   P                  V4       K›  MmV'       g   V	'       d   YereMZV'       gS   V	'       gK   W4^,           R  F  w  r«W¶8X  g   K  W¡9  pV'       g   K  T
p M 	  Rp\        P                  RV RV R	24       V P                  V4      pR
V9   d'   VP                  R
^4      w  rïV P                  V4      pMTpT p\        VWý4       V P                  VV4       Vf   EKp  X'       g   EK{  VP!                  V4       EK�  	  R# )a  
Tie the model weights. If `recompute_mapping=False` (default when called internally), it will rely on the
`model.all_tied_weights_keys` attribute, containing the `{target: source}` mapping for the tied params.
If `recompute_mapping=True`, it will re-check all internal submodels and their config to determine the params
that need to be tied. This is the default when `model.tie_weights()` is called on its own, outside of
`__init__`, and `from_pretrained`, in case the config values were changed somewhere.

Note that during `from_pretrained`, tying is *symmetric*: if the mapping says "tie target -> source" but
`source` is missing in the checkpoint while `target` exists, we *swap* source and target so we can still
tie everything to the parameter that actually exists.
Tr‰  NzDThe tied weights mapping and config for this model specifies to tie r}  z°, but both are present in the checkpoints with different values, so we will NOT tie them. You should update the config with `tie_word_embeddings=False` to silence this warning.FzYThis checkpoint seem corrupted. The tied weights mapping for this model specifies to tie zk, but both are absent from the checkpoint, and we could not find another related tied weight for those keysrY  )rŽ  rš  r°   r9  Ú	enumerater¯   ÚequalrŠ  rÎ  rñ  rl  Úget_parameter_or_bufferr¯  Úget_submoduler¦  Ú_adjust_biasÚdiscard)rš   r   r!  r   ÚiÚtarget_param_nameÚsource_param_nameÚremove_from_missingÚsource_is_thereÚtarget_is_thereÚtarget_backupÚsource_backupÚtarget_backup_is_thereÚsource_paramÚparent_namer^  r§  s   &&&              r’   rã  ÚPreTrainedModel.tie_weightse
  så  € ÷ !Ø×2Ñ2‰Ià×;Ñ;È$Ð;ÓOˆIä˜Ÿ™Ó*Ó+ˆ	Ü9BÀ9×9MÑ5ˆAÑ5Ð!àÒ'Ø&*Ð#Ø"3Ñ"G�Ø"3Ñ"G�÷ #§ô !Ÿ;š; t×'9Ñ'9Ð:KÓ'LÈd×N`ÑN`ÐarÓNs×tÒtÜŸ™ØbÐctÐbuÐuyØ0Ð1ð 2ðôð ×2Ñ2×6Ñ6Ð7HÔIá ð u÷ )¯_Ø;LÑ'8ç(·Ø8AÀaÅ%À'Ó8JÑ4˜ð )Ö=Ø5BÑ5VÐ2÷  6Ñ5Ø4AÐ 1Ù %ñ 9Kð  /4Ð+ÜŸ™ØwØ0Ð1°Ð6GÐ5Hð I_ð_ôð  ×7Ñ7Ð8IÓJˆLØÐ'Ô'Ø$5×$<Ñ$<¸SÀ!Ó$DÑ!�Ø×+Ñ+¨KÓ8‘à(�Ø�ä�F˜DÔ/Ø×Ñ˜f lÔ3àÕ'×,?Ò,?Ø×$Ñ$Ð%6×7ó} :Nr•   c                óÆ  € \        VR R4      e™   \        VR4      '       d‡   VP                  P                  p\        P
                  P                  VP                  P                  ^ V^ ,          VP                  P                  ^ ,          ,
          3R^ 4      VP                  n        \        VR4      '       d(   \        VR4      '       d   VP                  Vn
        R# R# R# )rÅ  NrÁ  ÚconstantÚout_featuresÚnum_embeddings)r[  rÀ   rÁ  r  r   Ú
functionalÚpadrÅ  Údatar8  r7  )rš   Úoutput_embeddingsrŽ  Úweight_shapes   &&& r’   r'  ÚPreTrainedModel._adjust_bias¸
  s½   € ÜÐ$ f¨dÓ3Ò?ÄGÐL]Ð_g×DhÒDhØ,×3Ñ3×9Ñ9ˆLÜ*,¯-©-×*;Ñ*;Ø!×&Ñ&×+Ñ+Ø�L •OÐ&7×&<Ñ&<×&BÑ&BÀ1Õ&EÕEÐFØØó	+Ð×"Ñ"Ô'ô Ð$ n×5Ò5¼'ÐBRÐTd×:eÒ:eØ-=×-LÑ-LÐÖ*ñ ;fÑ5r•   c                ób   <€ V ^8„  d   QhRS[ R,          RS[ R,          RS[RS[P                  /# )rŒ   Únew_num_tokensNÚpad_to_multiple_ofÚmean_resizingr�   )rÅ   r�   r   r&  )r�   r‘   s   "€r’   r“   rW  Ä
  s?   ø€ ÷ 9ñ 9á˜d�
ð9ñ   $�Jð9ñ ð	9ñ
 
�‰ñ9r•   c                ó0  € V P                  WV4      pVf   Vf   V# \        V R4      ;'       d    V P                  RJp\        4       '       dc   V'       g[   ^ RIpVP
                  P                  VP                  RR7      ;_uu_ 4        VP                  P                  ^ ,          pRRR4       MVP                  P                  ^ ,          pXV P                  P                  4       n        Wpn        V P                  4        V#   + '       g   i     LG; i)ad  
Resizes input token embeddings matrix of the model if `new_num_tokens != config.vocab_size`.

Takes care of tying weights embeddings afterwards if the model class has a `tie_weights()` method.

Arguments:
    new_num_tokens (`int`, *optional*):
        The new number of tokens in the embedding matrix. Increasing the size will add newly initialized
        vectors at the end. Reducing the size will remove vectors from the end. If not provided or `None`, just
        returns a pointer to the input tokens `torch.nn.Embedding` module of the model without doing anything.
    pad_to_multiple_of (`int`, *optional*):
        If set will pad the embedding matrix to a multiple of the provided value.If `new_num_tokens` is set to
        `None` will just pad the embedding to a multiple of `pad_to_multiple_of`.

        This is especially useful to enable the use of Tensor Cores on NVIDIA hardware with compute capability
        `>= 7.5` (Volta), or on TPUs which benefit from having sequence lengths be a multiple of 128. For more
        details about this, or help on choosing the correct value for resizing, refer to this guide:
        https://docs.nvidia.com/deeplearning/performance/dl-performance-matrix-multiplication/index.html#requirements-tc
    mean_resizing (`bool`):
        Whether to initialize the added embeddings from a multivariate normal distribution that has old embeddings' mean and
        covariance or to initialize them with a normal distribution that has a mean of zero and std equals `config.initializer_range`.

        Setting `mean_resizing` to `True` is useful when increasing the size of the embeddings of causal language models,
        where the generated tokens' probabilities won't be affected by the added embeddings because initializing the new embeddings with the
        old embeddings' mean will reduce the kl-divergence between the next token probability before and after adding the new embeddings.
        Refer to this article for more information: https://nlp.stanford.edu/~johnhew/vocab-expansion.html

Return:
    `torch.nn.Embedding`: Pointer to the input tokens Embeddings Module of the model.
Nr˜   ©Úmodifier_rank)Ú_resize_token_embeddingsrÀ   r˜   r-   rÝ  rà  ÚGatheredParametersrÁ  r  rÜ  rÍ  Ú
vocab_sizerã  )rš   r@  rA  rB  Úmodel_embedsr›   rÝ  rH  s   &&&&    r’   Úresize_token_embeddingsÚ'PreTrainedModel.resize_token_embeddingsÄ
  sì   € ðH ×4Ñ4°^ÐYfÓgˆØÒ!Ð&8Ò&@ØÐô ˜t ^Ó4×VÐV¸×9JÑ9JÐRVÐ9VˆÜ%×'Ò'·Ûà—‘×2Ñ2°<×3FÑ3FÐVZÐ2×[Õ[Ø)×0Ñ0×6Ñ6°qÕ9�
÷ \Ð[ð &×,Ñ,×2Ñ2°1Õ5ˆJð 4>ˆ�‰×#Ñ#Ó%Ô0Ø$Œð 	×ÑÔàÐ÷ \×[ús   Â
DÄD	c                ó`  € V P                  4       pV P                  WAW#4      p\        VR 4      '       d   VP                  p\	        WV4       VP
                  P                  pVP                  V4       V P                  V4       \        V R4      ;'       d    V P                  RJpVe�   \        4       '       dc   V'       g[   ^ RIp	V	P                  P                  VP
                  RR7      ;_uu_ 4        VP
                  P                  ^ ,          pRRR4       MVP
                  P                  ^ ,          pV P                  4       eÃ   V P                  4       p
\!        V
\"        P$                  P&                  4      '       d   V P                  W¡VR7      pMV P)                  W¡VR7      p\        V
R 4      '       d   V
P                  p\	        W¶4       V
P
                  P                  pVP                  V4       V P+                  V4       V P                  4       #   + '       g   i     Lô; i)Ú_hf_hookr˜   NrD  )rB  )r?  Ú_get_resized_embeddingsrÀ   rM  r|   rÁ  r£  rƒ  rI  r˜   r-   rÝ  rà  rG  r  rM  r‡  r¯   r   r&  Ú_get_resized_lm_headrQ  )rš   r@  rA  rB  Úold_embeddingsrP  r”  Úold_embeddings_requires_gradr›   rÝ  Úold_lm_headÚnew_lm_headÚold_lm_head_requires_grads   &&&&         r’   rF  Ú(PreTrainedModel._resize_token_embeddingsÿ
  sÏ  € Ø×2Ñ2Ó4ˆØ×5Ñ5ØÐ,>ó
ˆô �> :×.Ò.Ø!×*Ñ*ˆDÜ˜~Ô4Ø'5×'<Ñ'<×'JÑ'JÐ$Ø×%Ñ%Ð&BÔCØ×!Ñ! .Ô1Ü˜t ^Ó4×VÐV¸×9JÑ9JÐRVÐ9Vˆð Ò)Ü)×+Ò+·LÛ à—^‘^×6Ñ6°~×7LÑ7LÐ\`Ð6×aÕaØ%3×%:Ñ%:×%@Ñ%@ÀÕ%C�N÷ bÐað "0×!6Ñ!6×!<Ñ!<¸QÕ!?�ð ×%Ñ%Ó'Ò3Ø×4Ñ4Ó6ˆKÜ˜+¤u§x¡x×'9Ñ'9×:Ò:Ø"×:Ñ:¸;ÐfsÐ:Ót‘à"×7Ñ7¸ÐcpÐ7Óq�Ü�{ J×/Ò/Ø"×+Ñ+�Ü" ;Ô5Ø(3×(:Ñ(:×(HÑ(HÐ%Ø×&Ñ&Ð'@ÔAØ×&Ñ& {Ô3à×(Ñ(Ó*Ð*÷' b×aús   Ã5HÈH-	c          
      ó|   <€ V ^8„  d   QhRS[ P                  RS[R,          RS[R,          RS[RS[ P                  /# )rŒ   rP  r@  NrA  rB  r�   )r   r&  rÅ   r�   )r�   r‘   s   "€r’   r“   rW  &  sT   ø€ ÷ ]ñ ]áŸ™ð]ñ ˜d�
ð]ñ   $�Jð	]ñ
 ð]ñ 
�‰ñ]r•   c           	     óv
  € Vee   \        V\        4      '       g   \        RV R24      hVf   VP                  P                  ^ ,          pW#,           ^,
          V,          V,          pM\
        P                  RV R24       Vf   V# \        V R4      ;'       d    V P                  RJp\        4       '       db   V'       gZ   ^ RI
pVP                  P                  VP                  RR7      ;_uu_ 4        VP                  P                  4       w  rxRRR4       MVP                  P                  4       w  rxXV8X  d   \        4       '       g	   W!n        V# \        V\        P                   4      '       g;   \#        R\%        V4       R	\        P                    R
\        P                    R24      h\        P                   ! VXVP                  P&                  VP                  P(                  R7      p	W'8”  d   V'       g   V P+                  V	4       M¥W'8”  d    V'       d˜   \
        P-                  R4       W',
          p
\        4       '       dY   V'       gQ   ^ RI
pVP                  P                  VP                  .RR7      ;_uu_ 4        V P/                  WWz4       RRR4       MV P/                  WWz4       \1        Wr4      p\        4       '       d�   V'       gˆ   ^ RI
pVP                  V	P                  .pVP                  P                  V^ R7      ;_uu_ 4        VP                  P2                  RV1R3,          V	P                  P2                  RV1R3&   RRR4       M<VP                  P2                  RV1R3,          V	P                  P2                  RV1R3&   \        4       '       d¿   V'       g·   ^ RI
pVP                  V	P                  .pVP                  P                  V^ R7      ;_uu_ 4        V	P                  Vn        V	P                  P2                  P                  ^ ,          Vn        VP4                  e    V^,
          VP4                  8  d   RVn        RRR4       V# V	P                  P2                  VP                  n        V	P                  P2                  P                  ^ ,          Vn        VP4                  e    V^,
          VP4                  8  d   RVn        V#   + '       g   i     ELÅ; i  + '       g   i     ELY; i  + '       g   i     EL‡; i  + '       g   i     T# ; i)aâ  
Build a resized Embedding Module from a provided token Embedding Module. Increasing the size will add newly
initialized vectors at the end. Reducing the size will remove vectors from the end

Args:
    old_embeddings (`torch.nn.Embedding`):
        Old embeddings to be resized.
    new_num_tokens (`int`, *optional*):
        New number of tokens in the embedding matrix.

        Increasing the size will add newly initialized vectors at the end. Reducing the size will remove
        vectors from the end. If not provided or `None`, just returns a pointer to the input tokens
        `torch.nn.Embedding` module of the model without doing anything.
    pad_to_multiple_of (`int`, *optional*):
        If set will pad the embedding matrix to a multiple of the provided value. If `new_num_tokens` is set to
        `None` will just pad the embedding to a multiple of `pad_to_multiple_of`.

        This is especially useful to enable the use of Tensor Cores on NVIDIA hardware with compute capability
        `>= 7.5` (Volta), or on TPUs which benefit from having sequence lengths be a multiple of 128. For more
        details about this, or help on choosing the correct value for resizing, refer to this guide:
        https://docs.nvidia.com/deeplearning/performance/dl-performance-matrix-multiplication/index.html#requirements-tc
    mean_resizing (`bool`):
        Whether to initialize the added embeddings from a multivariate normal distribution that has old embeddings' mean and
        covariance or to initialize them with a normal distribution that has a mean of zero and std equals `config.initializer_range`.

        Setting `mean_resizing` to `True` is useful when increasing the size of the embeddings of causal language models,
        where the generated tokens' probabilities will not be affected by the added embeddings because initializing the new embeddings with the
        old embeddings' mean will reduce the kl-divergence between the next token probability before and after adding the new embeddings.
        Refer to this article for more information: https://nlp.stanford.edu/~johnhew/vocab-expansion.html


Return:
    `torch.nn.Embedding`: Pointer to the resized Embedding Module or the old Embedding Module if
    `new_num_tokens` is `None`
Nz5Asking to pad the embedding matrix to a multiple of `z@`, which is not and integer. Please make sure to pass an integerz�You are resizing the embedding layer without providing a `pad_to_multiple_of` parameter. This means that the new embedding dimension will be a.  . This might induce some performance reduction as *Tensor Cores* will not be available. For more details about this, or help on choosing the correct value for resizing, refer to this guide: https://docs.nvidia.com/deeplearning/performance/dl-performance-matrix-multiplication/index.html#requirements-tcr˜   rD  zOld embeddings are of type ú, which is not an instance of zj. You should either use a different resize function or make sure that `old_embeddings` are an instance of rY  r  zýThe new embeddings will be initialized from a multivariate normal distribution that has old embeddings' mean and covariance. As described in this article: https://nlp.stanford.edu/~johnhew/vocab-expansion.html. To disable this, use `mean_resizing=False`rÿ  )r‡  rÅ   rÛ   rÁ  r  rÎ  rÏ  rÀ   r˜   r-   rÝ  rà  rG  r4  r8  r   r&  ru  r‰  rä   r¦   ré  rà  Ú(_init_added_embeddings_weights_with_meanr  r;  rÙ  )rš   rP  r@  rA  rB  r›   rÝ  Úold_num_tokensÚold_embedding_dimrP  Úadded_num_tokensÚnÚparamss   &&&&&        r’   rN  Ú'PreTrainedModel._get_resized_embeddings&  s~  € ðV Ò)ÜÐ0´#×6Ò6Ü ØKÐL^ÐK_ð  ``ð  aóð ð Ò%Ø!/×!6Ñ!6×!<Ñ!<¸QÕ!?�Ø-ÕBÀQÕFÐK]Õ]ÐasÕs‰Nä�K‰Kð&Ø&4Ð%5ð 6DðDôð Ò!Ø!Ð!ä˜t ^Ó4×VÐV¸×9JÑ9JÐRVÐ9VˆÜ%×'Ò'·Ûà—‘×2Ñ2°>×3HÑ3HÐX\Ð2×]Õ]Ø4B×4IÑ4I×4NÑ4NÓ4PÑ1�÷ ^Ð]ð 1?×0EÑ0E×0JÑ0JÓ0LÑ-ˆNà˜^Ô+Ô4N×4PÒ4PØ,:Ô)Ø!Ð!ä˜.¬"¯,©,×7Ò7ÜØ-¬d°>Ó.BÐ-CÐCaÔbd×bnÑbnÐaoð pä—L‘L�> ð$óð ô ŸšØØØ!×(Ñ(×/Ñ/Ø ×'Ñ'×-Ñ-ô	
ˆð Ô*·=à×Ñ˜~Õ.àÔ,·ô ×Ñð=ôð  .Õ>ÐÜ)×+Ò+·LÛ à—^‘^×6Ñ6¸×8MÑ8MÐ7NÐ^bÐ6×cÕcØ×AÑAØ&¸ô÷ dÐcð
 ×=Ñ=Ø"°Nôô �Ó/ˆä%×'Ò'·Ûà$×+Ñ+¨^×-BÑ-BÐCˆFØ—‘×2Ñ2°6ÈÐ2×KÕKØ4B×4IÑ4I×4NÑ4NÈrÐPQÈrÐSTÈuÕ4U�×%Ñ%×*Ñ*¨2¨A¨2¨q¨5Ñ1÷ LÐKð 1?×0EÑ0E×0JÑ0JÈ2ÈAÈ2ÈqÈ5Õ0QˆN×!Ñ!×&Ñ& r¨ r¨1 uÑ-ô
 &×'Ò'·Ûà$×+Ñ+¨^×-BÑ-BÐCˆFØ—‘×2Ñ2°6ÈÐ2×KÕKØ(6×(=Ñ(=�Ô%Ø0>×0EÑ0E×0JÑ0J×0PÑ0PÐQRÕ0S�Ô-ð "×-Ñ-Ò9¸~ÐPQÕ?QÐUc×UoÑUoÔ>oØ15�NÔ.÷ Lð Ðð *8×)>Ñ)>×)CÑ)CˆN×!Ñ!Ô&Ø,:×,AÑ,A×,FÑ,F×,LÑ,LÈQÕ,OˆNÔ)Ø×)Ñ)Ò5¸>ÈAÕ;MÐQ_×QkÑQkÔ:kØ-1�Ô*àÐ÷w ^×]Ð]ú÷^ d×cÐcú÷$ L×KÐKú÷ LÖKð Ðús1   Ã5S+Ê	S?Ì=TÏ6A+T'Ó+S<	Ó?T	ÔT$	Ô'T8	c          
      ón   <€ V ^8„  d   QhRS[ P                  RS[R,          RS[RS[RS[ P                  /# )rŒ   rR  r@  NÚ
transposedrB  r�   )r   rÎ  rÅ   r�   )r�   r‘   s   "€r’   r“   rW  Å  sP   ø€ ÷ Añ Aá—Y‘YðAñ ˜d�
ðAñ ð	Añ
 ðAñ 
�‰ñAr•   c           
     óô  € Vf   V# \        V R4      ;'       d    V P                  RJp\        4       '       d’   V'       gŠ   ^ RIpVP                  P                  VP                  RR7      ;_uu_ 4        V'       g   VP                  P                  4       M'VP                  P                  4       P                  4       w  rxRRR4       MLV'       g   VP                  P                  4       M'VP                  P                  4       P                  4       w  rxXV8X  d   \        4       '       g	   W!n	        V# \        V\        P                  4      '       g;   \        R\        V4       R\        P                   R\        P                   R24      hV'       g   XV3MVX3p	VP                  RJp
\        P                  ! V	RV
R	VP                  P                   R
VP                  P"                  / pW'8”  d   V'       g   V P%                  V4       MøW'8”  dó   V'       dë   \&        P)                  R4       W',
          p\        4       '       d‘   V'       g‰   ^ RIpVP                  .pV
'       d   WÑP                  .,          pVP                  P                  VRR7      ;_uu_ 4        V P+                  WW‡WÃ4       V
'       d   V P-                  WV4       RRR4       M-V P+                  WW‡WÃ4       V
'       d   V P-                  WV4       \/        Wr4      p\        4       '       d}   V'       gu   ^ RIpVP                  VP                  VP                  VP                  .pVP                  P                  V^ R7      ;_uu_ 4        V P1                  W±WãV
4       RRR4       MV P1                  W±WãV
4       \3        VRR4       V#   + '       g   i     EL»; i  + '       g   i     LÝ; i  + '       g   i     LF; i)aî  
Build a resized Linear Module from a provided old Linear Module. Increasing the size will add newly initialized
vectors at the end. Reducing the size will remove vectors from the end

Args:
    old_lm_head (`torch.nn.Linear`):
        Old lm head liner layer to be resized.
    new_num_tokens (`int`, *optional*):
        New number of tokens in the linear matrix.

        Increasing the size will add newly initialized vectors at the end. Reducing the size will remove
        vectors from the end. If not provided or `None`, just returns a pointer to the input tokens
        `torch.nn.Linear` module of the model without doing anything. transposed (`bool`, *optional*, defaults
        to `False`): Whether `old_lm_head` is transposed or not. If True `old_lm_head.size()` is `lm_head_dim,
        vocab_size` else `vocab_size, lm_head_dim`.
    mean_resizing (`bool`):
        Whether to initialize the added embeddings from a multivariate normal distribution that has old embeddings' mean and
        covariance or to initialize them with a normal distribution that has a mean of zero and std equals `config.initializer_range`.

        Setting `mean_resizing` to `True` is useful when increasing the size of the embeddings of causal language models,
        where the generated tokens' probabilities will not be affected by the added embeddings because initializing the new embeddings with the
        old embeddings' mean will reduce the kl-divergence between the next token probability before and after adding the new embeddings.
        Refer to this article for more information: https://nlp.stanford.edu/~johnhew/vocab-expansion.html

Return:
    `torch.nn.Linear`: Pointer to the resized Linear Module or the old Linear Module if `new_num_tokens` is
    `None`
Nr˜   rD  z#Old language model head is of type rX  zg. You should either use a different resize function or make sure that `old_lm_head` are an instance of rY  rÅ  rä   r¦   a  The new lm_head weights will be initialized from a multivariate normal distribution that has old embeddings' mean and covariance. As described in this article: https://nlp.stanford.edu/~johnhew/vocab-expansion.html. To disable this, use `mean_resizing=False`rÆ  T)rÀ   r˜   r-   rÝ  rà  rG  rÁ  r4  rò   r7  r‡  r   rÎ  ru  r‰  rÅ  rä   r¦   ré  rÎ  rà  Ú%_init_added_lm_head_weights_with_meanÚ"_init_added_lm_head_bias_with_meanr  Ú!_copy_lm_head_original_to_resizedr¦  )rš   rR  r@  ra  rB  r›   rÝ  rZ  Úold_lm_head_dimÚnew_lm_head_shapeÚhas_new_lm_head_biasrS  r\  r^  Únum_tokens_to_copys   &&&&&          r’   rO  Ú$PreTrainedModel._get_resized_lm_headÅ  sn  € ðH Ò!ØÐä˜t ^Ó4×VÐV¸×9JÑ9JÐRVÐ9VˆÜ%×'Ò'·Ûà—‘×2Ñ2°;×3EÑ3EÐUYÐ2×ZÕZç5?�K×&Ñ&×+Ñ+Ô-À[×EWÑEW×EYÑEYÓE[×E`ÑE`ÓEbñ 0�÷ [ÐZ÷ 2<�×"Ñ"×'Ñ'Ô)À×ASÑAS×AUÑAUÓAW×A\ÑA\ÓA^ñ ,ˆNð ˜^Ô+Ô4N×4PÒ4PØ'5Ô$ØÐä˜+¤r§y¡y×1Ò1ÜØ5´d¸;Ó6GÐ5HÐHfÔgi×gpÑgpÐfqð rä—I‘I�;˜að!óð ÷ FP˜_¨nÑ=ÐVdÐfuÐUvÐØ*×/Ñ/°tÐ;Ðô —i’iØð
à%ð
ð ×%Ñ%×,Ñ,ð
ð ×$Ñ$×*Ñ*ñ	
ˆð Ô*·=à×Ñ˜{Õ+àÔ,·ô ×Ñð=ôð  .Õ>ÐÜ)×+Ò+·LÛ à%×,Ñ,Ð-�ß'Ø×/Ñ/Ð0Õ0�FØ—^‘^×6Ñ6°vÈTÐ6×RÕRØ×>Ñ>Ø#°/ÐScô÷ ,Ø×?Ñ?ÀÐZjÔk÷ SÐRð ×:Ñ:Ø¨oÐO_ô÷ (Ø×;Ñ;¸KÐVfÔgä  Ó@Ðä%×'Ò'·Ûà!×(Ñ(¨+×*:Ñ*:¸K×<NÑ<NÐP[×P`ÑP`ÐaˆFØ—‘×2Ñ2°6ÈÐ2×KÕKØ×6Ñ6ØÐ.@ÐNbô÷ LÐKð
 ×2Ñ2ØÐ*<ÐJ^ôô 	�Ð1°4Ô8ØÐ÷m [×ZÐZú÷p S×Rú÷( L×Kús%   Á4AO Ê#.OÎO'Ï O	ÏO$	Ï'O7	c                ó,  € VP                   P                  P                  \        P                  4      p\        P
                  ! V^ R7      pWV,
          pVP                  V,          V,          pRp	\        P                  P                  W˜,          4      P                  4       p
V
'       dŒ   \        P                  P                  P                  WiV,          R7      pVP                  V3R7      P                  VP                   P                  4      VP                   P                  RV,          R1R3&   R# VR,          P!                  V^4      P                  VP                   P                  4      VP                   P                  RV,          R1R3&   R# )	r   r	  ç•Ö&è.>)Úcovariance_matrix)Úsample_shapeNrÿ  rM  ©Nrÿ  )rÁ  r;  r:  r¯   rî   rÃ  ÚTr   Úpositive_definiteÚcheckró  ÚdistributionsÚmultivariate_normalÚMultivariateNormalÚsampler¦   r  )rš   rP  rP  rZ  r\  Úold_embeddings_weightÚmean_embeddingsÚold_centered_embeddingsÚ
covarianceÚepsilonÚis_covariance_psdÚdistributions   &&&&&       r’   rY  Ú8PreTrainedModel._init_added_embeddings_weights_with_meanH  sY  € ð !/× 5Ñ 5× :Ñ :× =Ñ =¼e¿m¹mÓ LÐÜŸ*š*Ð%:ÀÔCˆØ"7Õ"IÐØ,×.Ñ.Ð1HÕHÈ>ÕYˆ
ð ˆÜ'×9Ñ9×?Ñ?ÀÕ@TÓU×YÑYÓ[Ðßä ×.Ñ.×BÑB×UÑUØ¸ZÕ3Gð Vó ˆLð FR×EXÑEXØ.Ð0ð FYó Fç‰b�×&Ñ&×,Ñ,Ó-ð ×!Ñ!×&Ñ& rÐ,<Õ'<Ñ'>ÀÐ'AÓBð   Õ(×/Ñ/Ð0@À!ÓD×GÑGÈ×H]ÑH]×HcÑHcÓdð ×!Ñ!×&Ñ& rÐ,<Õ'<Ñ'>ÀÐ'AÓBr•   c                ó    <€ V ^8„  d   QhRS[ /# )rŒ   ra  rŽ   )r�   r‘   s   "€r’   r“   rW  a  s   ø€ ÷ @ñ @ñ ñ@r•   c                óÆ  € V'       d_   VP                   P                  P                  VP                   n        VP                   P                  P                  VP                   n        V P                  WWE4       V'       da   VP                   P                  P                  VP                   n        VP                   P                  P                  VP                   n        R # R # r—   )rÁ  r;  rp  rY  )rš   rR  rS  rf  rZ  r\  ra  s   &&&&&&&r’   rc  Ú5PreTrainedModel._init_added_lm_head_weights_with_meana  s¢   € ÷ à&1×&8Ñ&8×&=Ñ&=×&?Ñ&?ˆK×ÑÔ#Ø&1×&8Ñ&8×&=Ñ&=×&?Ñ&?ˆK×ÑÔ#ð 	×5Ñ5°kÐP^Ôqçà&1×&8Ñ&8×&=Ñ&=×&?Ñ&?ˆK×ÑÔ#Ø&1×&8Ñ&8×&=Ñ&=×&?Ñ&?ˆK×ÑÖ#ñ r•   c                ó~  € \         P                  ! VP                  P                  ^ \         P                  R7      p\         P
                  ! VP                  P                  ^ R7      P                  \         P                  4      pVP                  P                  RV,          R P                  VRV,          R7       R# )r   )r
  r¦   r	  Nrl  rÂ  rM  )r¯   rÃ  rÅ  r;  rî   rÄ  r:  rÔ  )rš   rR  rS  r\  Ú	bias_meanÚbias_stds   &&&&  r’   rd  Ú2PreTrainedModel._init_added_lm_head_bias_with_meanw  sƒ   € Ü—J’J˜{×/Ñ/×4Ñ4¸1ÄEÇMÁMÔRˆ	Ü—9’9˜[×-Ñ-×2Ñ2¸Ô;×>Ñ>¼u¿}¹}ÓMˆØ×Ñ×Ñ˜bÐ#3Õ3Ð5Ð6×>Ñ>ÀIÐSWÐZbÕSbÐ>Öcr•   c                ó|  € V'       g>   VP                   P                  R V1R3,          VP                   P                  R V1R3&   M<VP                   P                  RR V13,          VP                   P                  RR V13&   V'       d3   VP                  P                  R V VP                  P                  R V% R # R # ro  )rÁ  r;  rÅ  )rš   rS  rR  ri  ra  rh  s   &&&&&&r’   re  Ú1PreTrainedModel._copy_lm_head_original_to_resized|  sÀ   € ÷ Ø>I×>PÑ>P×>UÑ>UÐViÐWiÐViÐklÐVlÕ>mˆK×Ñ×#Ñ#Ð$7Ð%7Ð$7¸Ð$:Ò;à>I×>PÑ>P×>UÑ>UÐVWÐYlÐZlÐYlÐVlÕ>mˆK×Ñ×#Ñ# AÐ':Ð(:Ð':Ð$:Ñ;÷  Ø9D×9IÑ9I×9NÑ9NÐObÐPbÐ9cˆK×Ñ×!Ñ!Ð"5Ð#5Ò6ñ  r•   c                ó    <€ V ^8„  d   QhRS[ /# )rŒ   Únew_num_position_embeddingsrÄ   )r�   r‘   s   "€r’   r“   rW  ‰  s   ø€ ÷ 
ñ 
Ácñ 
r•   c           	     ó|   € \        R V P                   RV P                   RV P                  P                   R24      h)z4`resize_position_embeddings` is not implemented for úB`. To implement it, you should overwrite this method in the class ú in `modeling_ú.py`©rB  rC  r³   )rš   r‰  s   &&r’   Úresize_position_embeddingsÚ*PreTrainedModel.resize_position_embeddings‰  sH   € Ü!ØBÀ4Ç>Á>ÐBRð S2Ø26·.±.Ð1AÀÐPT×P^ÑP^×PiÑPiÐOjÐjnðpó
ð 	
r•   c                óh   <€ V ^8„  d   QhRS[ P                  S[S[ P                  ,          ,          /# r‹   )r   r&  rg  )r�   r‘   s   "€r’   r“   rW  �  s&   ø€ ÷ 
ñ 
©¯©¹¹b¿l¹lÕ8KÕ)Kñ 
r•   c           	     ó|   € \        R V P                   RV P                   RV P                  P                   R24      h)z1`get_position_embeddings` is not implemented for r‹  rŒ  r�  rŽ  r™   s   &r’   Úget_position_embeddingsÚ'PreTrainedModel.get_position_embeddings�  sH   € Ü!Ø?ÀÇÁÐ?Oð P2Ø26·.±.Ð1AÀÐPT×P^ÑP^×PiÑPiÐOjÐjnðpó
ð 	
r•   c                ó�   € \        4       \        P                  ! R4      8w  d   V P                  4        V P	                  RR7       R# )z“
Initialize and tie the weights if needed. If using a custom `PreTrainedModel`, you need to implement any
initialization logic in `_init_weights`.
r0  F)r!  N)rè   r¯   rä   r   rã  r™   s   &r’   r�  ÚPreTrainedModel.init_weights•  s5   € ô 6Ó7¼5¿<º<ÈÓ;OÔOà×#Ñ#Ô%à×Ñ¨5ÐÖ1r•   c                ó<  € V P                   '       g#   \        V P                  P                   R24      hVf   RR/p\        P
                  ! \        3/ VB pR\        P                  ! V P                  4      P                  9   pV'       g   V P                  RVR7       M;V P                  \        V P                  RR7      4       \        P                  R	4       V P                  R
8H  pT;'       g    \        V RR4      pV'       d   V P!                  4        R# R# )a¸  
Activates gradient checkpointing for the current model.

We pass the `__call__` method of the modules instead of `forward` because `__call__` attaches all the hooks of
the module. https://discuss.pytorch.org/t/any-different-between-model-input-and-model-forward-input/3690/2

Args:
    gradient_checkpointing_kwargs (dict, *optional*):
        Additional keyword arguments passed along to the `torch.utils.checkpoint.checkpoint` function.
z) does not support gradient checkpointing.NÚuse_reentrantFrG  T)ÚenableÚgradient_checkpointing_func©rG  áV  You are using an old version of the checkpointing format that is deprecated (We will also silently ignore `gradient_checkpointing_kwargs` in case you passed it).Please update to the new format on your modeling file. To use the new format, you need to completely remove the definition of the method `_set_gradient_checkpointing` in your model.rT  Ú_hf_peft_config_loaded)rÊ  rÛ   rC  r²   Ú	functoolsr
   r   rd  Ú	signatureÚ_set_gradient_checkpointingrò  ÚapplyrÎ  rñ  Úmain_input_namer[  r�  )rš   Úgradient_checkpointing_kwargsrš  Ú_is_using_old_formatÚneeds_embedding_gradsÚenable_input_gradss   &&    r’   rË  Ú-PreTrainedModel.gradient_checkpointing_enable¡  sþ   € ð ×3×3Ð3Ü §¡× 7Ñ 7Ð8Ð8aÐbÓcÐcà(Ò0Ø-<¸eÐ,DÐ)ä&/×&7Ò&7¼
Ñ&dÐFcÑ&dÐ#ð  '¬'×*;Ò*;¸D×<\Ñ<\Ó*]×*hÑ*hÑhÐç#Ø×,Ñ,°DÐVqÐ,Õrà�J‰J”w˜t×?Ñ?ÀtÔLÔMÜ�N‰NðHôð
 !%× 4Ñ 4¸Ñ CÐà2×dÐd´g¸dÐD\Ð^cÓ6dÐßð
 ×+Ñ+Ö-ñ r•   c                ó&   <€ V ^8„  d   QhRS[ RS[/# )rŒ   r™  rš  )r�   r   )r�   r‘   s   "€r’   r“   rW  Ë  s   ø€ ÷ ñ ±$ð Ñ\dñ r•   c                ó,  € R p\        V R4      '       d   W n        Wn        RpV P                  4        F3  p\        VR4      '       g   K  \	        VRV4       \	        VRV4       RpK5  	  V'       g#   \        V P                  P                   R24      hR# )FrÉ  TÚ_gradient_checkpointing_funczÂ is not compatible with gradient checkpointing. Make sure all the architecture support it by setting a boolean attribute `gradient_checkpointing` to modules of the model that uses checkpointing.N)rÀ   rª  rÉ  rg  r¦  rÛ   rC  r²   )rš   r™  rš  Úis_gradient_checkpointing_setrU  s   &&&  r’   r   Ú+PreTrainedModel._set_gradient_checkpointingË  sœ   € Ø(-Ð%ô �4Ð1×2Ò2Ø0KÔ-Ø*0Ô'Ø,0Ð)à—l‘l–nˆFÜ�vÐ7×8Ô8Ü˜Ð >Ð@[Ô\Ü˜Ð 8¸&ÔAØ04Ò-ñ	 %÷ -ÜØ—>‘>×*Ñ*Ð+ð ,]ð ]óð ñ -r•   c                óz  € V P                   '       d„   R\        P                  ! V P                  4      P                  9   pV'       g   V P                  RR7       M;\
        P                  R4       V P                  \        V P                  RR7      4       \        V RR4      '       d   V P                  4        R# R# )z;
Deactivates gradient checkpointing for the current model.
rG  F)r™  rœ  r›  r�  N)rÊ  rd  rŸ  r   rò  rÎ  rñ  r¡  r
   r[  r•  )rš   r¤  s   & r’   Úgradient_checkpointing_disableÚ.PreTrainedModel.gradient_checkpointing_disableá  s–   € ð ×/×/Ð/ð $+¬g×.?Ò.?À×@`Ñ@`Ó.a×.lÑ.lÑ#lÐ ß'Ø×0Ñ0¸Ð0Õ>ä—‘ðLôð —
‘
œ7 4×#CÑ#CÈ5ÔQÔRä�4Ð1°5×9Ò9Ø×,Ñ,Ö.ñ :r•   c                ó    <€ V ^8„  d   QhRS[ /# r‹   rŽ   )r�   r‘   s   "€r’   r“   rW  ö  s   ø€ ÷ nñ n©4ñ nr•   c                ó¢   € \         ;QJ d*    R V P                  4        4       F  '       g   K   R# 	  R# ! R V P                  4        4       4      # )zD
Whether gradient checkpointing is activated for this model or not.
c              3   ób   "  € T F%  p\        VR 4      ;'       d    VP                  x € K'  	  R# 5i)rÉ  N)rÀ   rÉ  )r  Úms   & r’   r  Ú<PreTrainedModel.is_gradient_checkpointing.<locals>.<genexpr>ú  s,   é € ÐmÑ^lÐYZ”7˜1Ð6Ó7×TÐT¸A×<TÑ<TÔTÓ^lùs   ‚/š/TF)r‹  rg  r™   s   &r’   Úis_gradient_checkpointingÚ)PreTrainedModel.is_gradient_checkpointingõ  sA   € ÷
 ‹sÑmÐ^b×^jÑ^jÔ^lÓm�sŒsÐmŠsÐmˆsÑmÐ^b×^jÑ^jÔ^lÓmÓmÐmr•   c                ó¾   <€ V ^8„  d   QhRS[ S[P                  ,          RS[RS[R,          RS[RS[S[ ,          RS[ R,          RS[ S[,          R,          R	S[R
S[/	# )rŒ   Úsave_directoryÚis_main_processrñ   NÚpush_to_hubÚmax_shard_sizer¬  r¼  Úsave_peft_formatÚsave_original_format)r­   rË   r-  r�   r®   rÅ   )r�   r‘   s   "€r’   r“   rW  ü  sŽ   ø€ ÷ \ñ \á™bŸk™kÕ)ð\ñ ð\ñ ˜4•Kð	\ñ
 ð\ñ ™c�	ð\ñ �t•ð\ñ ‘T�z˜DÕ ð\ñ ð\ñ #ñ\r•   c
           	     ó^  € Ve   WzR&   \        V RR4      p\        V RR4      pVRJ;'       d)    \        V\        4      ;'       d    VP                  4       pVe4   V'       g,   V'       g$   \	        RVP
                  P                   R24      hV P                  e   \        R4      '       g   \        R	4      h\        P                  P                  V4      '       d   \        P                  R
V R24       R# \        P                  ! VRR7       \        P                   ! V4      pV'       d�   V
P#                  RR4      pV
P#                  RVP%                  \        P                  P&                  4      R:,          4      pV
P#                  RR4      p\)        4       P*                  ! V3RR/V
B P,                  pV P/                  V4      p/ pVe   VP1                  V 4      w  ppRVR&   \3        V 4      pVP4                  p\7        V4      P%                  R4      ^,          VP8                  n        VP:                  P<                  P?                  R4      .VP8                  n         V PB                  e   \E        WV P8                  R7       V'       Ed   V'       g   VP8                  PG                  V4       V PI                  4       '       d   VPJ                  PG                  V4       V'       dÃ   \        PM                  R4       VPO                  VR7      pV'       d<   \        PM                  R4       / pVPQ                  4        F  w  ppVVRV 2&   K  	  TpV PS                  4       p\U        V4      ^8”  d   \	        R4      hV^ ,          pV PV                  V,          pVPG                  V4       Vf   VPY                  4       pRp\[        V R4      '       dˆ   \U        \]        V P^                  Pa                  4       4      4      ^8”  dW   RV P^                  Pa                  4       9   g    RV P^                  Pa                  4       9   d   Rp\b        Pd                  ! R4       \f        '       d7   \h        Pj                  Pl                  Pn                   F  w  ppV! V4      pK  	  V Pp                  e:   \U        V Pp                  4      ^ 8”  d    V Pp                   F  pVV9   g   K  VV K  	  V P                  e,   \s        W0Pt                  V Pv                  V P                  4      p\y        VV4      pV	'       d   V'       g   \{        VV4      pV'       g   \|        p\        VV4      pM\€        pVPƒ                  R R!4      Pƒ                  R"R#4      p \…        VV VR$7      p!Rp"V!P†                  '       d-   R%R&V P‰                  4       /V!PŠ                  CR'V!PŒ                  /p"\        PŽ                  ! V4       EF  p#\        P                  P‘                  VV#4      p$VPƒ                  R R(4      Pƒ                  R"R(4      p%V#Pƒ                  R R(4      Pƒ                  R"R(4      p&\’        P”                  ! R)4      p'V#P—                  V%4      '       g   K—  \        P                  P                  V$4      '       g   K¾  V#V!P˜                  9  g   KÑ  V'       g   KÛ  V'P›                  V&4      f   Kð  \        Pœ                  ! V$4       EK	  	  \ž        P                   ! V!P˜                  PQ                  4       R*R+7       F“  w  p(p)\        P                  P‘                  VV(4      p#/ p*V) FV  p+VP#                  V+4      p,V'       d(   V,P¢                  P¤                  R,8X  d   \§        VV+4      p,V,P©                  4       V*V+&   KX  	  \«        V*V#VR-7       ?*K•  	  V"f:   \        P                  P‘                  VV4      p-\        PM                  R.V- 24       M²\¬        p.\        P                  P‘                  V\        V.V4      4      p.\¯        V.R/R0R17      ;_uu_ 4       p/\°        P²                  ! V"^RR27      R3,           p0V/Pµ                  V04       RRR4       \        PM                  R4V R5\U        V!P˜                  4       R6V. R24       V'       da   \·        XV P¸                  VR77      p1V1P»                  \        P                  P‘                  VR84      4       V P½                  VVXXVXR97       R# R#   + '       g   i     L¬; i);a^  
Save a model and its configuration file to a directory, so that it can be re-loaded using the
[`~PreTrainedModel.from_pretrained`] class method.

Arguments:
    save_directory (`str` or `os.PathLike`):
        Directory to which to save. Will be created if it doesn't exist.
    is_main_process (`bool`, *optional*, defaults to `True`):
        Whether the process calling this is the main process or not. Useful when in distributed training like
        TPUs and need to call this function on all processes. In this case, set `is_main_process=True` only on
        the main process to avoid race conditions.
    state_dict (nested dictionary of `torch.Tensor`):
        The state dictionary of the model to save. Will default to `self.state_dict()`, but can be used to only
        save parts of the model or if special precautions need to be taken when recovering the state dictionary
        of a model (like when using model parallelism).
    push_to_hub (`bool`, *optional*, defaults to `False`):
        Whether or not to push your model to the Hugging Face model hub after saving it. You can specify the
        repository you want to push to with `repo_id` (will default to the name of `save_directory` in your
        namespace).
    max_shard_size (`int` or `str`, *optional*, defaults to `"50GB"`):
        The maximum size for a checkpoint before being sharded. Checkpoints shard will then be each of size
        lower than this size. If expressed as a string, needs to be digits followed by a unit (like `"5MB"`).

        <Tip warning={true}>

        If a single weight of the model is bigger than `max_shard_size`, it will be in its own checkpoint shard
        which will be bigger than `max_shard_size`.

        </Tip>

    variant (`str`, *optional*):
        If specified, weights are saved in the format model.<variant>.safetensors.
    token (`str` or `bool`, *optional*):
        The token to use as HTTP bearer authorization for remote files. If `True`, or not specified, will use
        the token generated when running `hf auth login` (stored in `~/.huggingface`).
    save_peft_format (`bool`, *optional*, defaults to `True`):
        For backward compatibility with PEFT library, in case adapter weights are attached to the model, all
        keys of the state dict of adapters needs to be prepended with `base_model.model`. Advanced users can
        disable this behaviours by setting `save_peft_format` to `False`.
    save_original_format (`bool`, *optional*, defaults to `True`):
        For backward compatibility with the previous versions of `transformers` you can save the checkpoint with
        its reverse mapping. The reverse mapping needs to exists even if the model was loaded from a None legacy
        checkpoint.
    kwargs (`dict[str, Any]`, *optional*):
        Additional key word arguments passed along to the [`~utils.PushToHubMixin.push_to_hub`] method.
Nr¼  r�  Fr˜   zThe model is quantized with z˜ and is not serializable - check out the warnings from the logger on the traceback to understand the reason why the quantized model is not serializable.z0.31.4z[Saving a model with tensor parallelism requires `huggingface_hub` version 0.31.4 or higher.zProvided path (z#) should be a directory, not a fileT)Úexist_okÚcommit_messageÚrepo_idÚ	create_prr¿  r2  r�   rY  ÚFSDP)rÜ  zhDetected adapters on the model, saving the model in the PEFT format, only adapter weights will be saved.)rñ   zƒTo match the expected format of the PEFT library, all keys of the state dict of adapters will be prepended with `base_model.model`.zbase_model.model.zßMultiple active adapters detected, saving multiple active adapters is not supported yet. You can save adapters separately one by one by iteratively calling `model.set_adapter(adapter_name)` then `model.save_pretrained(...)`Úhf_device_maprâ   Údiskz}Attempting to save a model with offloaded modules. Ensure that unallocated cpu memory exceeds the `shard_size` (50GB default)z.binz{suffix}.binr/  z{suffix}.safetensors)Úfilename_patternr»  ÚmetadataÚtotal_parametersÚ
weight_maprÀ  z(.*?)-\d{5}-of-\d{5}zWriting model shards)Údescr0  )rÇ  zModel weights saved in Úwr  r  )ÚindentÚ	sort_keysÚ
z:The model is bigger than the maximum size per checkpoint (z) and is going to be split in z^ checkpoint shards. You can find where each parameters has been saved in the index located at )r¼  z	README.md)rÀ  r¼  rÂ  rM  )_r[  r‡  rR   Úis_serializablerÛ   Úquantization_configÚquant_methodÚ_tp_sizers   r  rË   r  rÌ  rÎ  ÚerrorÚmakedirsr  rl  r  Úseprq   Úcreate_reporÁ  Ú_get_files_timestampsÚget_state_dict_and_metadataÚunwrap_modelr¦   r­   rÜ  rC  r²   ÚremoveprefixÚarchitecturesÚ_auto_classr'   Úsave_pretrainedr  r‚  rÏ  Úget_adapter_state_dictr9  Úactive_adaptersrí   Úpeft_configrñ   rÀ   rf  rÄ  rì   r¼  r½  ÚIS_SAGEMAKER_MP_POST_1_10ÚsmpÚstateÚmodule_managerÚtranslate_functionsr•  rE   rŒ  Ú_device_meshrŸ  r%   rY   r°  rV   rº  r   rÐ  r1  rÇ  Útensor_to_filenameÚlistdirrË  rƒ  r  r  Úfilename_to_tensorsÚ	fullmatchr“  rj   Útqdmrä   r‰  r5   Ú
contiguousÚsafe_save_filerX   r   ÚjsonÚdumpsÚwritero   rÑ  ÚsaveÚ_upload_modified_files)2rš   r¸  r¹  rñ   rº  r»  r¬  r¼  r¼  r½  rÉ  r�  r˜   Úquantization_serializableÚsave_directory_pathrÀ  rÁ  rÂ  Úfiles_timestampsrÇ  Úmodel_to_saver¦   Úpeft_state_dictr  rG  Úactive_adapterÚcurrent_peft_configÚis_offloadedÚ	smp_to_hfrp  Ú
ignore_keyr«  rÆ  Ústate_dict_splitÚindexrÓ  Úfull_filenameÚweights_no_suffixÚfilename_no_suffixÚregÚ
shard_fileÚtensor_namesÚshard_state_dictÚtensor_namerã   Úpath_to_weightsÚsave_index_filerF  ÚcontentÚ
model_cards2   &&&&&&&&&&,                                       r’   rÝ  ÚPreTrainedModel.save_pretrainedü  sÞ  € ðv ÒØ#�7‰Oä!(¨Ð/GÈÓ!OÐä˜t ^°TÓ:ˆà Ð$×qÐq¬°LÄ+Ó)N×qÐqÐS_×SoÑSoÓSqð 	"ð Ò#×,B×KdÜØ.¨|×/OÑ/O×/\Ñ/\Ð.]ð ^uð uóð ð �=‰=Ò$Ô-PÐQY×-ZÒ-ZÜØmóð ô �7‰7�>‰>˜.×)Ò)Ü�L‰L˜?¨>Ð*:Ð:]Ð^Ô_Ùä
�Š�N¨TÕ2Ü Ÿiši¨Ó7ÐçØ#ŸZ™ZÐ(8¸$Ó?ˆNØ—j‘j Ð,?×,EÑ,EÄbÇgÁgÇkÁkÓ,RÐSUÕ,VÓWˆGØŸ
™
 ;°Ó6ˆIÜ“h×*Ò*¨7ÑL¸TÐLÀVÑL×TÑTˆGØ#×9Ñ9¸.ÓIÐàˆØÒ#Ø#/×#KÑ#KÈDÓ#QÑ ˆJ˜Ø!ˆ�Ñô % TÓ*ˆð ×#Ñ#ˆÜ%(¨£Z×%5Ñ%5°cÓ%:¸1Õ%=ˆ×ÑÔ"ð /<×.EÑ.E×.NÑ.N×.[Ñ.[Ð\bÓ.cÐ-dˆ×ÑÔ*ð ×ÑÒ'Ü˜t¸D¿K¹KÕH÷ ˆ?ß)Ø×$Ñ$×4Ñ4°^ÔDØ× Ñ ×"Ò"Ø×/Ñ/×?Ñ?ÀÔOç%Ü—‘Ø~ôð +×AÑAÈZÐAÓX�
ç#Ü—K‘Kð ^ôð ')�OØ&0×&6Ñ&6Ö&8™
˜˜UØEJ˜Ð*;¸C¸5Ð(AÓBñ '9à!0�Jà!%×!5Ñ!5Ó!7�ä�~Ó&¨Ô*Ü$ðuóð ð "0°Õ!2�à&*×&6Ñ&6°~Õ&FÐ#Ø#×3Ñ3°NÔCð ÒØ&×1Ñ1Ó3ˆJð ˆä�D˜/×*Ò*Ü”C˜×*Ñ*×1Ñ1Ó3Ó4Ó5¸Ô9Ø˜$×,Ñ,×3Ñ3Ó5Ô5¸À4×CUÑCU×C\ÑC\ÓC^Ô9^àˆLÜ�MŠMð:ô÷ %Ó$Ü #§	¡	× 8Ñ 8× LÔ L‘�	˜1Ù& zÓ2’
ñ !Mð ×'Ñ'Ò3¼¸D×<XÑ<XÓ8YÐ\]Ô8]Ø"×:Ô:�
Ø Ö+Ø" :Ò.ñ ;ð
 �=‰=Ò$Ü3°JÇÁÈt×O`ÑO`Ðbf×boÑboÓpˆJô 9¸À]ÓSˆ
÷  ×(>Ü1°-ÀÓLˆJ÷ &Ü,ˆLÜ'¨°gÓ>‰Lä4ˆLà'×/Ñ/°¸ÓG×OÑOÐP^Ð`vÓwÐÜ=ØÐ)9È.ô
Ðð ˆØ×&×&Ð&àÐ/°×1DÑ1DÓ1FÐdÐJZ×JcÑJcÐdØÐ.×AÑAðˆEô Ÿ
š
 >×2ˆHÜŸG™GŸL™L¨¸ÓBˆMð !-× 4Ñ 4°V¸RÓ @× HÑ HÈÐY[Ó \Ðð "*×!1Ñ!1°&¸"Ó!=×!EÑ!EÀnÐVXÓ!YÐÜ—*’*Ð4Ó5ˆCð ×#Ñ#Ð$5×6Ô6Ü—G‘G—N‘N =×1Ô1ØÐ$4×$HÑ$HÖHß#‘OØ—M‘MÐ"4Ó5ÔAä—	’	˜-×(ñ# 3ô( )0¯ªØ×0Ñ0×6Ñ6Ó8Ð?U÷)
Ñ$ˆJ˜ô —w‘w—|‘| N°JÓ?ˆHØ!ÐÛ+�à#Ÿ™¨Ó4�÷
   F§M¡M×$6Ñ$6¸&Ô$@Ü5°mÀ[ÓQ�Fð 17×0AÑ0AÓ0CÐ  Ó-ñ  ,ô  Ð+¨XÀÕIâ ñ/)
ð2 Š=Ü Ÿg™gŸl™l¨>¸<ÓHˆOÜ�K‰KÐ1°/Ð1BÐCÕDä5ˆOÜ Ÿg™gŸl™l¨>¼<ÈÐY`Ó;aÓbˆOä�o s°W×=Õ=ÀÜŸ*š* U°1ÀÔEÈÕL�Ø—‘˜Ô ÷ >ô �K‰KØLÈ^ÐL\ð ]ÜÐ 0× DÑ DÓEÐFð G$Ø$3Ð#4°Að7ô÷ ä2°7¸D¿O¹OÐSXÔYˆJð �O‰OœBŸG™GŸL™L¨¸ÓEÔFà×'Ñ'ØØØ Ø-ØØ#ð (ö ñ ÷ >×=ús   á2däd,	c                ó  <€ V P                   e   V P                   M. pVP                  R. 4      p\        V\        4      '       d   V.pV F  pWS9  g   K  VP	                  V4       K  	  V'       d   W2R&   \
        SV `  ! V/ VB # )NrÏ  )rÑ  rÍ   r‡  r­   ri  rb  rº  )rš   rÈ  rÉ  rÏ  Útags_kwargsrÒ  rC  s   &*,   €r’   rº  ÚPreTrainedModel.push_to_hub  sw   ø€ à"&§/¡/Ò"=ˆt�ŠÀ2ˆà—j‘j ¨Ó,ˆÜ�k¤3×'Ò'Ø&˜-ˆKãˆCØŽØ—‘˜CÖ ñ ÷ Ø!�6‰NÜ‰wÒ" DÐ3¨FÑ3Ð3r•   c                ó¦   € \        R V P                  4        4       4      pV'       d)   \        R V P                  4        4       4      pW#,           pV# )a¼  
Get the memory footprint of a model. This will return the memory footprint of the current model in bytes.
Useful to benchmark the memory footprint of the current model and design some tests. Solution inspired from the
PyTorch discussions: https://discuss.pytorch.org/t/gpu-memory-that-model-uses/56822/2

Arguments:
    return_buffers (`bool`, *optional*, defaults to `True`):
        Whether to return the size of the buffer tensors in the computation of the memory footprint. Buffers
        are tensors that do not require gradients and not registered as parameters. E.g. mean and std in batch
        norm layers. Please see: https://discuss.pytorch.org/t/what-pytorch-means-by-buffers/120266/2
c              3   ój   "  € T F)  qP                  4       VP                  4       ,          x € K+  	  R # 5ir—   ©rN  rQ  rî  s   & r’   r  Ú7PreTrainedModel.get_memory_footprint.<locals>.<genexpr>6  s(   é € ÐYÑGX¸e—.‘.Ó" U×%7Ñ%7Ó%9×9Ò9ÓGXùó   ‚13c              3   ój   "  € T F)  qP                  4       VP                  4       ,          x € K+  	  R # 5ir—   r  )r  Úbufs   & r’   r  r  8  s%   é € ÐYÉ.À3Ÿ<™<›>¨C×,<Ñ,<Ó,>×>Ò>Ë.ùr  )Úsumrò  rô  )rš   Úreturn_buffersÚmemÚmem_bufss   &&  r’   Úget_memory_footprintÚ$PreTrainedModel.get_memory_footprint*  s@   € ô ÑYÀtÇÁÔGXÓYÓYˆßÜÑYÈ$Ï,É,Ì.ÓYÓYˆHØ•.ˆCØˆ
r•   c                óÞ  <€ \        V R R4      \        P                  8X  d€   ^ RIHp \
        SV `  ! V/ VB  V P                  4        FS  p\        WC4      '       g   K  \        V4      ^ 8”  d   V^ ,          pMVP                  RR4      pVP                  V4       KU  	  V # \        V R R4      \        P                  8X  d   \        V RR4      '       d   \        R4      h\
        SV `  ! V/ VB # )Úquantization_methodN©Ú	HQQLinearrä   r  Úis_loaded_in_8bitFzœCalling `cuda()` is not supported for `8-bit` quantized models.  Please use the model as it is, since the model has already been set to the correct devices.)r[  r{   ÚHQQÚhqq.core.quantizer  rb  r  rg  r‡  rí   rÍ   ÚBITS_AND_BYTESrÛ   )rš   rÈ  rÉ  r  rU  rä   rC  s   &*,   €r’   r  ÚPreTrainedModel.cuda<  sØ   ø€ ä�4Ð.°Ó5Ô9K×9OÑ9OÔOÝ3ô ‰GŠL˜$Ð) &Ò)ØŸ,™,ž.�Ü˜f×0Ô0Ü˜4“y 1”}Ø!% a¥™à!'§¡¨H°fÓ!=˜Ø—K‘K Ö'ñ )ð ˆKô �4Ð.°Ó5Ô9K×9ZÑ9ZÔZÜ�tÐ0°%×8Ò8Ü ðsóð ô ‰wŠ|˜TÐ, VÑ,Ð,r•   c                ó”  <€ R V9   pV'       g.   V F'  p\        V\        P                  4      '       g   K%  Rp M	  \        V RR4      \        P
                  8X  d–   ^ RIHp \        S	V `$  ! V/ VB  V P                  4        Fi  p\        We4      '       g   K  RV9   d   VR,          pM	V^ ,          pR V9   d   VR ,          pMV'       d   XpMRpVe   W†n        VP                  V4       Kk  	  V # V'       d,   \        V RR4      \        P                  8X  d   \        R4      h\        V RR4      \        P                  8X  dD   V'       d   \        R4      h\        V RR	4      '       d   \!        R
4      '       g   \        R4      hM3\        V RR4      \        P"                  8X  d   V'       d   \        R4      h\        S	V `$  ! V/ VB # )r¦   Tr  Nr  rä   zBCasting a Quark quantized model to a new `dtype` is not supported.z­You cannot cast a bitsandbytes model in a new `dtype`. Make sure to load the model using `from_pretrained` using the desired `dtype` by passing the correct `dtype` argument.r   Fz0.48zsYou need to install `pip install bitsandbytes>=0.48.0` if you want to move a 8-bit model across devices using to().z¥You cannot cast a GPTQ model in a new `dtype`. Make sure to load the model using `from_pretrained` using the desired `dtype` by passing the correct `dtype` argument.)r‡  r¯   r¦   r[  r{   r!  r"  r  rb  r:  rg  Úcompute_dtyper  ÚQUARKrÛ   r#  rd   ÚGPTQ)
rš   rÈ  rÉ  Údtype_present_in_argsÚargr  rU  rä   r¦   rC  s
   &*,      €r’   r:  ÚPreTrainedModel.toV  s®  ø€ ð !(¨6Ñ 1Ðç$Û�Ü˜c¤5§;¡;×/Ô/Ø,0Ð)Ùñ ô
 �4Ð.°Ó5Ô9K×9OÑ9OÔOÝ3ô ‰GŠJ˜Ð' Ò'ØŸ,™,ž.�Ü˜f×0Ô0Ø 6Ô)Ø!'¨Õ!1™à!% a¥˜Ø &Ô(Ø & w¥™ß.Ø #™à $˜ð Ò(Ø/4Ô,Ø—K‘K Ö'ñ# )ð$ ˆKç ¤W¨TÐ3HÈ$Ó%OÔSe×SkÑSkÔ%kÜÐaÓbÐbô �4Ð.°Ó5Ô9K×9ZÑ9ZÔZß$Ü ðPóð ô
 �tÐ0°%×8Ò8ÔAZÐ[a×AbÒAbÜ ð Jóð øô �TÐ0°$Ó7Ô;M×;RÑ;RÔRß$Ü ðHóð ô ‰wŠz˜4Ð* 6Ñ*Ð*r•   c                ó\   <€ \        V R R4      '       d   \        R4      h\        SV `  ! V!  # )r›   FzŽ`.half()` is not supported for quantized model. Please use the model as it is, since the model has already been casted to the correct `dtype`.)r[  rÛ   rb  Úhalf©rš   rÈ  rC  s   &*€r’   r-  ÚPreTrainedModel.half“  s6   ø€ ä�4˜¨×/Ò/ÜðIóð ô
 ‘7’< Ñ&Ð&r•   c                ó\   <€ \        V R R4      '       d   \        R4      h\        SV `  ! V!  # )r›   Fz�`.float()` is not supported for quantized model. Please use the model as it is, since the model has already been casted to the correct `dtype`.)r[  rÛ   rb  Úfloatr.  s   &*€r’   r1  ÚPreTrainedModel.float�  s6   ø€ ä�4˜¨×/Ò/ÜðIóð ô
 ‘7’= $Ñ'Ð'r•   c          	      óT   <€ V ^8„  d   QhRS[ P                  RS[RS[RS[R,          /# )rŒ   r¦   r›   rÕ   rq  N)r¯   r¦   r�   )r�   r‘   s   "€r’   r“   rW  ¨  s7   ø€ ÷ ñ Ù—K‘KðÙ/3ðÙIMðÙbfÐimÕbmñr•   c                ó¬  € \        WP                  4      \        P                  ! 4       \	        4       .pV'       d   VP                  \        4       4       \        4       '       d¶   ^ RIpV'       gq   V'       gi   \        P                  R4       VP                  \        P                  ! 4       VP                  P                  \        4       R7      \!        4       .4       V# V'       d0   VP                  \"        P$                  ! R4      \'        4       .4       V# VP                  \"        P$                  ! R4      \        P(                  ! 4       .4       V# )r   NrÙ  rÚ  r0  )rà   r²   rÞ  Úno_tie_weightsrO   ri  r<   r-   rÝ  rÎ  rÏ  r\  rß  rà  rá  r+   rÖ   r¯   rä   rÒ   Úmeta_device_safe_creation_ops)rg  r¦   r›   rÕ   rq  rä  rÝ  s   &&&&&  r’   Úget_init_contextÚ PreTrainedModel.get_init_context§  sü   € ô
 +¨5·,±,Ó?Ä×ATÒATÓAVÔXeÓXgÐhˆçØ× Ñ Ô!6Ó!8Ô9Ü%×'Ò'Û÷  ×(:Ü—‘Ð^Ô_Ø×$Ñ$ä×,Ò,Ó.Ø!Ÿ™×+Ñ+Ô@PÓ@RÐ+ÓSÜ'Ó)ðôð Ð÷ Ø×$Ñ$¤e§l¢l°6Ó&:Ô<OÓ<QÐ%RÔSð Ðð × Ñ ¤%§,¢,¨vÓ"6¼×8ZÒ8ZÓ8\Ð!]Ô^àÐr•   c                ó:   <€ V ^8„  d   QhRS[ P                  RS[/# )rŒ   r¦   r�   )r¯   r¦   r®   )r�   r‘   s   "€r’   r“   rW  Ç  s   ø€ ÷ ñ ¡U§[¡[ð ±Tñ r•   c                óª  € / pV P                   eS   V\        P                  8X  d>   VP                  \        P                  V P                   \        P                  4      4       V P                  ec   V\        P                  \        P                  39   d>   VP                  \        P                  V P                  \        P                  4      4       V# )z\Create the dtype_plan describing modules/parameters that should use the `keep_in_fp32` flag.)	r�  r¯   r  rœ  r®   Úfromkeysrî   r�  r  )rš   r¦   r§   s   && r’   Ú_get_dtype_planÚPreTrainedModel._get_dtype_planÇ  s‘   € àˆ
ð
 ×%Ñ%Ò1°e¼u¿}¹}Ô6LØ×ÑœdŸm™m¨D×,FÑ,FÌÏÉÓVÔWð ×,Ñ,Ò8¸UÄuÇ}Á}ÔV[×VdÑVdÐFeÔ=eØ×ÑœdŸm™m¨D×,MÑ,MÌuÏ}É}Ó]Ô^àÐr•   c                ó.   <€ V ^8„  d   QhRS[ R,          /# )rŒ   rw  N)r]   )r�   r‘   s   "€r’   r“   rW  ×  s   ø€ ÷ "&ñ "&¹,ÈÕ:Mñ "&r•   c                ó8  € V'       d„   \        4       '       g   \        R4      h^RIHp V! 4        VeH   \	        V\
        4      '       d2   W n        VP                  V 4       VP                  V 4       RV n	        R# RV n	        RV n        R# RV n	        RV n        R# )aw  
Set whether or not to use the `kernels` library to kernelize some layers of the model.
Args:
    use_kernels (`bool`):
        Whether or not to use the `kernels` library to kernelize some layers of the model.
    kernel_config (`KernelConfig`, *optional*):
        The kernel configuration to use to kernelize the model. If `None`, the default kernel mapping will be used.
zk`use_kernels=True` requires kernels>=0.9.0. Please install the latest version with `pip install -U kernels`)Ú$register_kernel_mapping_transformersNTF)
rf   rÛ   Úintegrations.hub_kernelsr@  r‡  r]   rw  Úsanitize_kernel_mappingÚcreate_compatible_mappingÚuse_kernels)rš   rD  rw  r@  s   &&& r’   Úset_use_kernelsÚPreTrainedModel.set_use_kernels×  s‘   € ÷ Ü'×)Ò)Ü ð Bóð õ Wá0Ô2àÒ(¬Z¸Ä|×-TÒ-Tà%2Ô"ð ×5Ñ5°dÔ;ð ×7Ñ7¸Ô=Ø#'�Ö ð $(�Ô Ø%)�Ö"à$ˆDÔØ!%ˆDÖr•   rÜ  r¸  r¡   r¹  r»  r¼  r½  r¾  r    rª   Úfusion_configr¬   c                ó¨  <€ V ^8„  d   QhRS[ S[,          RS[S[P                  ,          R,          RS[S[,          S[P                  ,          R,          RS[S[P                  ,          R,          RS[RS[RS[R	S[S[,          R,          R
S[RS[R,          RS[RS[S[S[S[S[S[3,          ,          3,          R,          RS[R,          RS[/# )rŒ   rg  rž   NrÜ  r¸  r¡   r¹  r»  r¼  r½  r    rª   rG  r¬   r�   )	r‰  rƒ   r­   rË   r-  r    r�   r®   r   )r�   r‘   s   "€r’   r“   rW  ü  s  ø€ ÷ Sñ SÙÑ-Õ.ðSá'*©R¯[©[Õ'8¸4Õ'?ðSñ !¡3Õ&©¯©Õ4°tÕ;ð	Sñ
 ™Ÿ™Õ$ tÕ+ðSñ "&ðSñ ðSñ ðSñ ‘T�z˜DÕ ðSñ ðSñ  �ðSñ ðSñ ™C¡©©S±#¨X­Õ!6Ð6Õ7¸$Õ>ðSñ ˜T•kðSñ  
%ñ!Sr•   c               ó*  € VP                  RR4      pVP                  RR4      pVP                  RR4      pVP                  RR4      pVP                  RR4      pVP                  RR4      pVP                  R	R4      pVP                  R
R4      pVP                  RR4      pVP                  RR4      pVP                  RR4      pVP                  RR4      pVP                  RR4      pVP                  RR4      pVP                  RR4      pVP                  RR4      pVP                  R/ 4      ;'       g    / P                  4       pVP                  RR4      p VP                  RR4      p!VP                  RR4      p"VP                  RR4      p#VP                  RR4      p$VP                  RR4      p%VP                  RR4      p&VP                  RR4      p'VP                  RR4      p(VP                  RR4      p)VP                  R R4      p*VP                  R!R4      p+V%e   V#f   R"p#RK F  p,VP                  V,R4      p-K  	  Ve	   Ve   TMTpVf   R"p\        4       '       d   V'       g   R#pR$VR%VRVR&VR'VR(VRV/p./ V.CR)V/Cp/Ve   Vf   V"e   \        R*4      hVR"8X  dE   \	        \
        P                  P                  R+R,4      4      '       d   \        P                  R-4       V#f   V$e   \        V#V$V&VR.7      w  pp&p$V"e   \        4       '       g   \        R/4      hVf   / p\        VV/3/ VB w  p0pp\        V4      pR0R1R2R3R4V/p1Ve   VV1R5&   \        V\        4      '       g|   Ve   TMTp2V P                   p3V3f   \        V P"                   R624      hV3P$                  ! V23R7R#RV"RVRV/V.BVB w  pp4RV49   d   V4P                  R4       V4P                  RV4      pM%\        P&                  ! V4      pTp4\)        VRV4      pVV/R)&   R8V9   d   VP                  R84      Vn        R9V9   d   VP                  R94      Vn        \/        VVVV
V14      w  p5ppV"'       dQ   V5e   \        R:4      hVe>   \        V\0        4      '       d   R;VP3                  4       9   g   R;V9   d   \5        R<4      hV*e    V)'       g   \        P7                  R=4       R#p)\9        VVV"V	V/V1V P;                  4       \)        VR>R4      VR?7	      w  p6p7V5RJp8\=        VV6VV7WúV54      w  ppV"'       dP   ^R@IH p9 \B        PD                  ! RA4      ;_uu_ 4        V ! V4      p:RRR4       V9! V6^ ,          R#X:VRB7      RC,          pWn#        Ve   \        P&                  ! V4      Vn$        \)        VRDR4      pVe   ^REI%H&p; V;! WV4       V*e   V)'       d   ^RFI'H(p< V<! WV*4       V PS                  VV8\T        V(4      p=\        P&                  ! V4      p\W        V=4      ;_uu_ 4        V ! V.VO5/ V4B p>\Y        V>4       V5e   V5P[                  V>VVV6V)RG7       RRR4       X>P]                  V4      p?\_        V>V+V54      p@\`        '       d   V&e   \c        V>V#V%V&V$4      p>Ve   \e        V>VVV54      p\g        VVV7VVVVV?V5V&V
X@V	V.VRH7      pAV Pi                  V>VV6VA4      w  pBpCV Pk                  V>VAVB4      pBV>Pm                  4        V>Po                  V)V*4       V>Pq                  4       '       d7   \s        V>RI4      '       d%   V"'       g   V>Pt                  ! V!VVV3/ V.BRV'/BVB  Ve8   \w        \y        VP3                  4       4      4      ^8”  d   \{        V>V5VVXCV4       V5e   V5V>n>        V5P                  V>4       V0e   Ve   VVR'&   V>P�                  V0V XAVRJ7      pBV'       d   V>XBPƒ                  4       3# V>#   + '       g   i     ELœ; i  + '       g   i     EL»; i)La´:  
Instantiate a pretrained pytorch model from a pre-trained model configuration.

The model is set in evaluation mode by default using `model.eval()` (Dropout modules are deactivated). To train
the model, you should first set it back in training mode with `model.train()`.

The warning *Weights from XXX not initialized from pretrained model* means that the weights of XXX do not come
pretrained with the rest of the model. It is up to you to train those weights with a downstream fine-tuning
task.

The warning *Weights from XXX not used in YYY* means that the layer XXX is not used by YYY, therefore those
weights are discarded.

Parameters:
    pretrained_model_name_or_path (`str` or `os.PathLike`, *optional*):
        Can be either:

            - A string, the *model id* of a pretrained model hosted inside a model repo on huggingface.co.
            - A path to a *directory* containing model weights saved using
              [`~PreTrainedModel.save_pretrained`], e.g., `./my_model_directory/`.
            - `None` if you are both providing the configuration and state dictionary (resp. with keyword
              arguments `config` and `state_dict`).
    model_args (sequence of positional arguments, *optional*):
        All remaining positional arguments will be passed to the underlying model's `__init__` method.
    config (`Union[PreTrainedConfig, str, os.PathLike]`, *optional*):
        Can be either:

            - an instance of a class derived from [`PreTrainedConfig`],
            - a string or path valid as input to [`~PreTrainedConfig.from_pretrained`].

        Configuration for the model to use instead of an automatically loaded configuration. Configuration can
        be automatically loaded when:

            - The model is a model provided by the library (loaded with the *model id* string of a pretrained
              model).
            - The model was saved using [`~PreTrainedModel.save_pretrained`] and is reloaded by supplying the
              save directory.
            - The model is loaded by supplying a local directory as `pretrained_model_name_or_path` and a
              configuration JSON file named *config.json* is found in the directory.
    state_dict (`dict[str, torch.Tensor]`, *optional*):
        A state dictionary to use instead of a state dictionary loaded from saved weights file.

        This option can be used if you want to create a model from a pretrained configuration but load your own
        weights. In this case though, you should check if using [`~PreTrainedModel.save_pretrained`] and
        [`~PreTrainedModel.from_pretrained`] is not a simpler option.
    cache_dir (`Union[str, os.PathLike]`, *optional*):
        Path to a directory in which a downloaded pretrained model configuration should be cached if the
        standard cache should not be used.
    ignore_mismatched_sizes (`bool`, *optional*, defaults to `False`):
        Whether or not to raise an error if some of the weights from the checkpoint do not have the same size
        as the weights of the model (if for instance, you are instantiating a model with 10 labels from a
        checkpoint with 3 labels).
    force_download (`bool`, *optional*, defaults to `False`):
        Whether or not to force the (re-)download of the model weights and configuration files, overriding the
        cached versions if they exist.
    proxies (`dict[str, str]`, *optional*):
        A dictionary of proxy servers to use by protocol or endpoint, e.g., `{'http': 'foo.bar:3128',
        'http://hostname': 'foo.bar:4012'}`. The proxies are used on each request.
    output_loading_info(`bool`, *optional*, defaults to `False`):
        Whether or not to also return a dictionary containing missing keys, unexpected keys and error messages.
    local_files_only(`bool`, *optional*, defaults to `False`):
        Whether or not to only look at local files (i.e., do not try to download the model).
    token (`str` or `bool`, *optional*):
        The token to use as HTTP bearer authorization for remote files. If `True`, or not specified, will use
        the token generated when running `hf auth login` (stored in `~/.huggingface`).
    revision (`str`, *optional*, defaults to `"main"`):
        The specific model version to use. It can be a branch name, a tag name, or a commit id, since we use a
        git-based system for storing models and other artifacts on huggingface.co, so `revision` can be any
        identifier allowed by git.

        <Tip>

        To test a pull request you made on the Hub, you can pass `revision="refs/pr/<pr_number>"`.

        </Tip>
    attn_implementation (`str`, *optional*):
        The attention implementation to use in the model (if relevant). Can be any of
            - `"eager"` (manual implementation of the attention)
            - `"sdpa"` (using [`F.scaled_dot_product_attention`](https://pytorch.org/docs/master/generated/torch.nn.functional.scaled_dot_product_attention.html))
            - `"flash_attention_2"` (using [Dao-AILab/flash-attention](https://github.com/Dao-AILab/flash-attention))
            - `"flash_attention_3"` (using [Dao-AILab/flash-attention/hopper](https://github.com/Dao-AILab/flash-attention/tree/main/hopper))
            - `"flash_attention_4"` (using [Dao-AILab/flash-attention/flash_attn/cute](https://github.com/Dao-AILab/flash-attention/tree/main/flash_attn/cute)).
        By default, if available, SDPA will be used. The default is otherwise the manual `"eager"` implementation.

        Accept HF kernel references in the form:
          <namespace>/<repo_name>[@<revision>][:<kernel_name>]

        - <namespace> and <repo_name> are any non-"/" and non-":" sequences.
        - "@<revision>" is optional (branch, tag, or commit-ish), e.g. "@main", "@v1.2.0", "@abc123".
        - ":<kernel_name>" is optional and selects a function inside the kernel repo.
        - Both options can appear together and in this order only: @revision first, then :kernel_name.
        - We intentionally allow a leading "<wrapper>|" prefix (e.g., "flash|...") because the code
          strips it before loading; '|' is not excluded in the character classes here.

        Examples that match:
          "org/model"
          "org/model@main"
          "org/model:custom_kernel"
          "org/model@v1.2.3:custom_kernel"
    experts_implementation (`str`, *optional*):
        The experts implementation to use in the model (if relevant). Can be any of:

        - `"eager"` (sequential implementation of the experts matrix multiplications).
        - `"batched_mm"` (using [`torch.bmm`](https://pytorch.org/docs/stable/generated/torch.bmm.html)).
        - `"grouped_mm"` (using [`torch.nn.functional.grouped_mm`](https://docs.pytorch.org/docs/main/generated/torch.nn.functional.grouped_mm.html)).

        By default, if the model supports it, `"grouped_mm"` will be used. The default is otherwise the manual `"eager"` implementation.

    > Parameters for big model inference

    dtype (`str` or `torch.dtype`, *optional*, defaults to `"auto"`):
        Override the default `torch_dtype` and load the model under a specific `dtype`. The different options
        are:

        1. `torch.float16` or `torch.bfloat16` or `torch.float`: load in a specified
          `dtype`, ignoring the model's `config.dtype` if one exists. If not specified
          - the model will get loaded in `torch.float` (fp32).

        2. `"auto"` - A `dtype` or `torch_dtype` entry in the `config.json` file of the model will be
          attempted to be used. If this entry isn't found then next check the `dtype` of the first weight in
          the checkpoint that's of a floating point type and use that as `dtype`. This will load the model
          using the `dtype` it was saved in at the end of the training. It can't be used as an indicator of how
          the model was trained. Since it could be trained in one of half precision dtypes, but saved in fp32.

        3. A string that is a valid `torch.dtype`. E.g. "float32" loads the model in `torch.float32`, "float16" loads in `torch.float16` etc.

        <Tip>

        For some models the `dtype` they were trained in is unknown - you may try to check the model's paper or
        reach out to the authors and ask them to add this information to the model's card and to insert the
        `dtype` or `torch_dtype` entry in `config.json` on the hub.

        </Tip>

    device_map (`str` or `dict[str, Union[int, str, torch.device]]` or `int` or `torch.device`, *optional*):
        A map that specifies where each submodule should go. It doesn't need to be refined to each
        parameter/buffer name, once a given module name is inside, every submodule of it will be sent to the
        same device. If we only pass the device (*e.g.*, `"cpu"`, `"cuda:1"`, `"mps"`, or a GPU ordinal rank
        like `1`) on which the model will be allocated, the device map will map the entire model to this
        device. Passing `device_map = 0` means put the whole model on GPU 0.

        To have Accelerate compute the most optimized `device_map` automatically, set `device_map="auto"`. For
        more information about each option see [designing a device
        map](https://hf.co/docs/accelerate/main/en/usage_guides/big_modeling#designing-a-device-map).
    max_memory (`Dict`, *optional*):
        A dictionary device identifier to maximum memory if using `device_map`. Will default to the maximum memory available for each
        GPU and the available CPU RAM if unset.
    tp_plan (`Optional[Union[dict, str]]`, *optional*):
        A torch tensor parallel plan, see [here](https://pytorch.org/tutorials/intermediate/TP_tutorial.html). Use `tp_plan="auto"` to
        use the predefined plan based on the model. If it's a dict, then it should match between module names and desired layout.
        Note that if you use it, you should launch your script accordingly with `torchrun [args] script.py`. This will be much
        faster than using a `device_map`, but has limitations.
    tp_size (`str`, *optional*):
        A torch tensor parallel degree. If not provided would default to world size.
    device_mesh (`torch.distributed.DeviceMesh`, *optional*):
        A torch device mesh. If not provided would default to world size. Used only for tensor parallel for now.
        If provided, it has to contain dimension named `"tp"` in case it's > 1 dimensional, this dimension will be used for tensor parallelism
    offload_folder (`str` or `os.PathLike`, *optional*):
        If the `device_map` contains any value `"disk"`, the folder where we will offload weights.
    offload_buffers (`bool`, *optional*):
        Whether or not to offload the buffers with the model parameters.
    quantization_config (`Union[QuantizationConfigMixin,Dict]`, *optional*):
        A dictionary of configuration parameters or a QuantizationConfigMixin object for quantization (e.g
        bitsandbytes, gptq).
    subfolder (`str`, *optional*, defaults to `""`):
        In case the relevant files are located inside a subfolder of the model repo on huggingface.co, you can
        specify the folder name here.
    variant (`str`, *optional*):
        If specified load weights from `variant` filename, *e.g.* pytorch_model.<variant>.bin.
    use_safetensors (`bool`, *optional*, defaults to `None`):
        Whether or not to use `safetensors` checkpoints. Defaults to `None`. If not specified and `safetensors`
        is not installed, it will be set to `False`.
    weights_only (`bool`, *optional*, defaults to `True`):
        Indicates whether unpickler should be restricted to loading only tensors, primitive types,
        dictionaries and any types added via torch.serialization.add_safe_globals().
        When set to False, we can load wrapper tensor subclass weights.
    disable_mmap (`bool`, *optional*):
        Whether to disable memory mapping when loading safetensors checkpoints. When `None` (default),
        it is auto-detected to `True` when the checkpoint lives on an `hf-mount` FUSE filesystem
        (used by HF Spaces/Endpoints), where mmap + parallel page-faults can deadlock. When `True`,
        files are read fully into memory and parsed with `safetensors.torch.load`. When `False`, the
        default memory-mapped loader is always used.
    fusion_config (`dict[str, bool | dict[str, Any]]`, *optional*):
        Optional fusion configuration applied before model instantiation. Each key enables a fusion family and
        its value can either be `True` to enable that fusion with default options or a dictionary of
        family-specific options. For example, `{"patch_embeddings": True}` enables patch embedding fusion.
        This should only be used as an inference optimization, as it can slightly change outputs. If omitted,
        `from_pretrained()` falls back to `config.fusion_config` when available. Refer to the fusion mapping
        guide in `docs/source/en/fusion_mapping.md` for more details.
    key_mapping (`dict[str, str], *optional*):
        A potential mapping of the weight names if using a model on the Hub which is compatible to a Transformers
        architecture, but was not converted accordingly.
    kwargs (remaining dictionary of keyword arguments, *optional*):
        Can be used to update the configuration object (after it being loaded) and initiate the model (e.g.,
        `output_attentions=True`). Behaves differently depending on whether a `config` is provided or
        automatically loaded:

            - If a configuration is provided with `config`, `**kwargs` will be directly passed to the
              underlying model's `__init__` method (we assume all relevant updates to the configuration have
              already been done)
            - If a configuration is not provided, `kwargs` will be first passed to the configuration class
              initialization function ([`~PreTrainedConfig.from_pretrained`]). Each key of `kwargs` that
              corresponds to a configuration attribute will be used to override said attribute with the
              supplied `kwargs` value. Remaining keys that do not correspond to any configuration attribute
              will be passed to the underlying model's `__init__` function.

<Tip>

Activate the special ["offline-mode"](https://huggingface.co/transformers/installation.html#offline-mode) to
use this method in a firewalled environment.

</Tip>

Examples:

```python
>>> from transformers import BertConfig, BertModel

>>> # Download model and configuration from huggingface.co and cache.
>>> model = BertModel.from_pretrained("google-bert/bert-base-uncased")
>>> # Model was saved using *save_pretrained('./test/saved_model/')* (for example purposes, not runnable).
>>> model = BertModel.from_pretrained("./test/saved_model/")
>>> # Update configuration during loading.
>>> model = BertModel.from_pretrained("google-bert/bert-base-uncased", output_attentions=True)
>>> assert model.config.output_attentions == True
```
rñ   Nrº  r¶  Úoutput_loading_infoFÚ_from_pipelineÚ
_from_autor¦   rÖ  r£   Ú
max_memoryÚoffload_folderr¥   rÐ  r¿  rÀ  rÄ  r¬  Úadapter_kwargsÚadapter_namerÌ  r‚  r²  r¯  Útp_sizer­  r©   Útrust_remote_coderq  rD  rw  Úkey_mappingrÞ  Tr¸  r¹  r»  r¼  r½  rÁ  zq`state_dict` cannot be passed together with a model name or a `gguf_file`. Use one of the two loading strategies.Ú
WORLD_SIZEr�   a  You've set device_map=`auto` while triggering a distributed run with torchrun. This might lead to unexpected behavior. If your plan is to load the model on each device, you should set device_map={: PartialState().process_index} where PartialState comes from accelerate library)rQ  r©   r£   zIaccelerate is required when loading a GGUF file `pip install accelerate`.Ú	file_typer~  r3  ÚpytorchÚfrom_auto_classÚusing_pipelinezN does not define `config_class`; pass an explicit config to `from_pretrained`.Úreturn_unused_kwargsr×  rØ  zÂYou cannot combine Quantization and loading a model from a GGUF file, try again by making sure you did not passed a `quantization_config` or that you did not load a quantized model from the Hub.rÅ  zxOne or more modules is configured to be mapped to disk. Disk offload is not supported for models loaded from GGUF files.zœA kernel_config was provided but use_kernels is False; setting use_kernels=True automatically. To suppress this warning, explicitly set use_kernels to True.Útransformers_weights)	rž   r¬  r²  r    rŸ   r³  r´  rµ  r¶  )Úload_gguf_checkpointr0  )Úreturn_tensorsÚmodel_to_loadrÖ  rd  rG  )Úregister_fusion_patches)Ú(register_kernel_replacements_and_fusions)r~  r¦   r£   rÙ  rD  )rž   r¡   r¢   r£   r¤   r¥   r¦   r§   r˜   r©   rª   r«   r    rŸ   r¬   Úadjust_generation_fn)rP  Úload_configrO  )ÚmirrorÚ
_fast_initÚlow_cpu_mem_usageÚfrom_tfÚ	from_flaxÚoffload_state_dict)Brl  r—  r   rÛ   rÅ   rË   rÌ   rÍ   rÎ  rÏ  rF   rc   r?   r2   r‡  r    ra  r²   Úfrom_pretrainedÚdeepcopyr[  ry  r}  rS   r®   rì   r�  rà  rÚ  r´  rå  Úmodeling_gguf_pytorch_utilsr[  r¯   rä   rv  rG  Úfusion_mappingr^  rA  r_  r7  rÕ   r\   rP   Úpreprocess_modelr<  r!   r¿   rD   r/   rˆ   Ú_load_pretrained_modelÚ_finalize_model_loadingÚevalrE  r  rÀ   r`  rí   rf  r1   r˜   Úpostprocess_modelÚload_adapterÚto_dict)Drg  rž   rÜ  r¸  r¡   r¹  r»  r¼  r½  r    rª   rG  r¬   Ú
model_argsrÉ  rñ   rº  r¶  rJ  Úfrom_pipelinerW  r¦   rÖ  r£   rM  rN  r¥   rÐ  r¿  rÁ  r¬  rO  rP  r‚  r²  r¯  rQ  r­  r©   rR  rq  rD  rw  rS  r^  rp  rŸ   Údownload_kwargs_with_commitÚ_adapter_model_pathr³  Úconfig_pathra  Úmodel_kwargsr˜   rÙ  r¢   r›   r[  Údummy_modelr^  r_  Úmodel_init_contextr~  r§   Úweight_conversionsra  Úloading_infoÚdisk_offload_indexsD   &&$$$$$$$$$$$*,                                                     r’   rh  ÚPreTrainedModel.from_pretrainedû  sê  € ðj —Z‘Z ¨dÓ3ˆ
Ø—*‘*˜Y¨Ó-ˆØ—Z‘Z ¨dÓ3ˆ
Ø$Ÿj™jÐ)>ÀÓFÐØŸ
™
Ð#3°TÓ:ˆØ Ÿ*™* \°5Ó9ˆØ—
‘
˜7 DÓ)ˆØ—j‘j °Ó5ˆØ—Z‘Z ¨dÓ3ˆ
Ø—Z‘Z ¨dÓ3ˆ
ØŸ™Ð$4°dÓ;ˆØ Ÿ*™*Ð%6¸Ó>ˆØ$Ÿj™jÐ)>ÀÓEÐØ—J‘J˜{¨BÓ/ˆ	Ø—j‘j °Ó6ˆØ—*‘*˜Y¨Ó-ˆØ Ÿ*™*Ð%5°rÓ:×@Ð@¸b×FÑFÓHˆØ—z‘z .°)Ó<ˆØ"ŸJ™JÐ':¸DÓAÐØ—J‘J˜{¨DÓ1ˆ	Ø—*‘*˜Y¨Ó-ˆØ—*‘*˜Y¨Ó-ˆØ06·
±
Ð;OÐQUÓ0VÐØ—j‘j °Ó5ˆØ"ŸJ™JÐ':¸DÓAÐØ"ŸJ™JÐ':¸EÓBÐØ—j‘j °Ó6ˆØŸ
™
 ?°DÓ9ˆØ—j‘j °Ó5ˆàÒ)¨gªoØˆGó pˆDØ—
‘
˜4 Ó&ŠAñ pð Ò"Ø"Ò.‘E°KˆEØŠ=ØˆEä×Ò×%5Ø#Ðð ˜Ø˜nØ�wØÐ 0Ø�UØ˜Ø˜ð
ˆð 'V¨Ð&U¸-ÈÑ&UÐ#àÒ!Ð'DÒ'PÐT]ÒTiÜð Dóð ð ˜Ô¤C¬¯
©
¯©°|ÀSÓ(I×$JÒ$JÜ�K‰Kðcôð Ò 'Ò"5Ü/LØ °kÈjô0Ñ,ˆJ˜ Wð Ò Ô)@×)BÒ)BÜÐhÓiÐiàÒ!ØˆNäM`Ø)Ø'ñN
ð ñN
ÑJÐÐ:¸Nô
 .¨jÓ9ˆ
à! 7¨K¸ÐDUÐWfÐgˆ
ØÒ$Ø+8ˆJÐ'Ñ(ô ˜&Ô"2×3Ò3Ø$*Ò$6™&Ð<YˆKØ×+Ñ+ˆLØÒ#Ü Ø—|‘|�nÐ$rÐsóð ð $0×#?Ò#?Øñ$à%)ð$ð $ð$ð +ð	$ð
  -ð$ð "ð$ð ñ$Ñ ˆF�Lð ˜lÔ*Ø× Ñ  Ô-Ø&×*Ñ*¨>¸;ÓG‰Kä—]’] 6Ó*ˆFØ!ˆLÜ! &¨.¸+ÓFˆKà5@Ð# MÑ2ð ! FÔ*Ø*0¯*©*Ð5JÓ*KˆFÔ'à# vÔ-Ø-3¯Z©ZÐ8PÓ-QˆFÔ*ä+;ØÐ'¨°\À:ó,
Ñ(ˆ�f˜j÷ ØÒ'Ü ð Yóð ð Ò%Ü˜J¬×-Ò-°&¸J×<MÑ<MÓ<OÔ2OÐTZÐ^hÔThä"ð.óð ð
 Ò$¯[Ü×Ñð oôð ˆKä-KØ*GØØØ+Ø7Ø!Ø×-Ñ-Ó/Ü+2°6Ð;QÐSWÓ+XØ!ô
.
Ñ*ÐÐ*ð $¨4Ð/ˆô #ØÐ# VÐ-=¸zÐYeó
‰ˆ�÷ ÝIô —’˜f×%Õ%Ù! &›k�÷ &ñ .Ø  Õ#°DÈÐafôàõˆJð <Ôð Ò$Ü#'§=¢=°Ó#?ˆFÔ ô   ¨¸Ó>ˆØÒ$Ý?á# C°Ô?ð Ò$¯ÝZá4°SÀ-ÔPà ×1Ñ1°%¸ÔGYÐ[lÓmÐä—’˜vÓ&ˆÜÐ/×0Õ0Ù˜Ð< Ò<¨|Ñ<ˆEÜ" 5Ô)àÒ'Ø×-Ñ-ØØØ)Ø%5Ø +ð .ô ÷ 1ð ×*Ñ*¨5Ó1ˆ
ô :¸%ÀÈlÓ[Ðç'Ó'¨KÒ,CÜ$ U¨GÐ5GÈÐV]Ó^ˆEð Ò!Ü(¨°
¸JÈÓUˆJô *Ø*GØ$;Ø-Ø!Ø .Ø+ØØ!Ø%Ø#Ø%Ø-Ø+Ø+Ø%ô
ˆð" ,/×+EÑ+EÀeÈZÐYiÐkvÓ+wÑ(ˆÐ(Ø×2Ñ2°5¸+À|ÓTˆØ�
‰
ŒØ×Ñ˜k¨=Ô9ð ×Ñ×Ò¤G¨EÐ3I×$JÒ$J×S\Ø×&Ò&Ø!ØØØ-ñ	ð
 "ñð #4ñð òð Ò!¤c¬#¨j×.?Ñ.?Ó.AÓ*BÓ&CÀaÔ&GÜ  |°ZÀÐQcÐetÔuàÒ#Ø!-ˆEÔØ×*Ñ*Øôð Ò*ØÒ Ø*/�˜wÑ'Ø ×-Ñ-Ø#Ø)Ø'Ø-ð	 .ó ˆL÷ Ø˜,×.Ñ.Ó0Ð0Ð0Øˆ÷e &×%Ð%ú÷: 1×0Ð0ús   Õ	_-Ø2`ß-_>	à`	c                óœ   <€ V ^8„  d   QhRRRS[ R,          RS[S[,          R,          RS[RS[S[,          R,          RS[S[S[ 3,          /# )	rŒ   r~  r„   rñ   NrÙ  ra  Úexpected_keysr�   )r®   r°   r­   rˆ   rg  rw   )r�   r‘   s   "€r’   r“   rW    sn   ø€ ÷ c0ñ c0Ø ðc0á˜4•Kðc0ñ ™s�) dÕ*ðc0ñ )ð	c0ñ
 ™C•y 4Õ'ðc0ñ 
Ñ ¡$Ð&Õ	'ñc0r•   c           
     ó  € VP                   pVP                  pVRJ;'       d8    VP                  P                  \        P
                  \        P                  09   pVf(   \        V P                  4       P                  4       4      MTp\        P                  \        P                  8¼  d   \        V\        V RR4      4       RpVP                   ec   RVP                   P#                  4       9   dD   \%        V VP&                  VVP                   VP(                  VP*                  VP,                  4      pVP                   e5   V'       g-   \/        VP                   V4      p	\1        W	VP                   4       . p
\3        4       '       d}   V'       gu   Vf@   / pV F5  pVP5                  \7        VRVP8                  VP:                  R7      4       K7  	  Tp\=        WV4      w  r­\?        VV
\A        4       \A        4       / R7      pWè3# \A        4       pVe   TpEM&Veã   V^ ,          PC                  R4      '       dÅ   VfÁ   / pV F·  pVP:                  '       g   \E        V4      '       dH   \G        VR4      ;_uu_ 4       pVP5                  \I        VPK                  4       4      4       RRR4       Km  \M        VR	RR
7      pVPO                  V4       VP                  4        F  pVPQ                  V4      VV&   K  	  K¹  	  M@Ve2   / pV F(  pVP5                  \7        WÃP:                  R7      4       K*  	  M\S        R4      h\U        V VVV PV                  VR7      w  rèV F  pVPY                  RRR4       K  	  Wè3#   + '       g   i     LÜ; i)zzPerform the actual loading of some checkpoints into a `model`, by reading them from disk and dispatching them accordingly.NrŒ  rÅ  râ   )r,  rª   r¬   )r   Ú
error_msgsÚunexpected_keysÚmismatched_keysÚconversion_errorsr/  r1  r2  )r3  rä   )r¬   z5Neither a state dict nor checkpoint files were found.)r~  rñ   ra  r¯  r}  )-r˜   r›   rÐ  rÑ  r{   r!  r'  r°   rñ   r;  rÎ  Úlevelrj   ÚWARNINGrH   r[  r£   rì   r0   r¤   r¢   r¦   r«   r3   Úcaching_allocator_warmupr-   rœ  rJ  rª   r¬   r6   rw   rf  r6  r)  r   r7  r8  r   rk  r<  rÛ   r$   r¯  Ú__exit__)r~  rñ   rÙ  ra  r€  r˜   r›   Úis_hqq_or_quarkr}  Úexpanded_device_mapr‚  Úmerged_state_dictÚ	ckpt_filer   r|  Úall_pointerÚfilerC  Úfile_pointerrD  s   &&&&&               r’   rm  Ú&PreTrainedModel._load_pretrained_model  s3  € ð #×/Ñ/ˆØ"×/Ñ/ˆØ&¨dÐ2÷ 
ð 
°|×7WÑ7W×7dÑ7dÜ×"Ñ"Ü×$Ñ$ði
ñ 8
ˆð <IÒ;Pœ˜U×-Ñ-Ó/×4Ñ4Ó6Ô7ÐVcˆä�<‰<œ7Ÿ?™?Ô*Ü˜=¬'°%¸ÀTÓ*JÔKð "Ðà×!Ñ!Ò-°&¸K×<RÑ<R×<YÑ<YÓ<[Ô2[Ü!8ØØ×/Ñ/Ø Ø×&Ñ&Ø×,Ñ,Ø×!Ñ!Ø×*Ñ*ó"Ðð ×!Ñ!Ò-·oÜ"3°K×4JÑ4JÈMÓ"ZÐÜ$ UÀ×AYÑAYÔZàˆ
ä%×'Ò'·ØÒ!Ø$&Ð!Û!1�IØ%×,Ñ,Ü'Ø%Ø).Ø)4×)AÑ)AØ)4×)AÑ)Aô	öñ "2ð /�
Ü'HÈÐ\gÓ'hÑ$ˆJä,Ø)Ø%Ü #£Ü #£Ø"$ôˆLðT Ð/Ð/ôE ›%ˆKØÒ%Ø$.Ò!Ø!Ò-Ð2BÀ1Õ2E×2NÑ2NÈ~×2^Ò2^ÐcmÒcuØ$&Ð!Û,�DØ"×/×/Ð/´?À4×3HÒ3HÜ! $¨×-Ô-°Ø-×4Ñ4Ô5EÀcÇhÁhÃjÓ5QÔR÷ .á Ü#,¨T¸TÈ%Ô#P�LØ—O‘O LÔ1Ø)×.Ñ.Ö0˜Ø/;×/EÑ/EÀaÓ/HÐ)¨!Ó,ó 1ò -ð "Ò-Ø$&Ð!Û!1�IØ%×,Ñ,¬_¸Y×UmÑUmÔ-nÖoò "2ô !Ð!XÓYÐYä/SØØ,Ø'ØŸ™Ø#5ô0Ñ,ˆLó !�Ø—
‘
˜4  tÖ,ñ !ð Ð/Ð/÷7 .×-ús   É;)M8Í8Nc                ó,   <€ V ^8„  d   QhRS[ RS[RS[/# )rŒ   ra  r|  r�   )rˆ   rw   )r�   r‘   s   "€r’   r“   rW  x  s%   ø€ ÷ $ñ $Ù/ð$Ù?Pð$á	ñ$r•   c           
     óæ  €  V P                  V4       V P                  VP                  4       VP                  VP                  VP
                  4       V P                  VP                  4       V P                  VP                  RR7       V P                  V4       \        V VP                  VP                  V\        R7       V#   \        T TP                  TP                  T\        R7       i ; i)a  Perform all post processing operations after having loaded some checkpoints into a model, such as moving
missing keys from meta device to their expected device, reinitializing missing weights according to proper
distributions, tying the weights and logging the loading report.F)r   r!  )r~  rž   r¡   r|  rÎ  )Ú mark_tied_weights_as_initializedÚ&_move_missing_keys_from_meta_to_deviceÚmissing_and_mismatchedr£   r©   r˜   Ú_initialize_missing_keysr›   rã  r   Ú#_adjust_missing_and_unexpected_keysrx   rž   r¡   rÎ  )r~  ra  r|  s   &&&r’   rn  Ú'PreTrainedModel._finalize_model_loadingw  sà   € ð	à×2Ñ2°<Ô@ð ×8Ñ8Ø×3Ñ3Ó5Ø×&Ñ&Ø×'Ñ'Ø×(Ñ(ô	ð ×*Ñ*¨;×+CÑ+CÔDð ×Ñ¨<×+DÑ+DÐX]ÐÔ^ð ×5Ñ5°lÔCä!ØØ.9×.WÑ.WØ(3×(KÑ(KØ)Üõð Ðøô "ØØ.9×.WÑ.WØ(3×(KÑ(KØ)Üöús   ‚BC Ã*C0c           
     ó   € V Uu0 uF&  pR P                  VP                  R 4      RR 4      kK(  	  ppTP                  V Uu0 uFW  p\        V4      ^ 8”  g   K  VR,          P	                  4       '       g   K4  R P                  VP                  R 4      RR 4      kKY  	  up4      p. pV P                  4        Fˆ  w  rxV'       d"   V P                   R 2p	VP                  V	4      pMAV'       d:   \        V4      ^ 8”  d   R P                  V P                  V.4      MV P                  pWu9   g   Kw  VP                  V4       KŠ  	  V# u upi u upi )rY  NrM  rP  )	rË  r  Úunionrí   ÚisdigitrZ  rê  rÚ  ri  )
rš   r’  Ú
add_prefixÚremove_prefixr  Úmodule_keysÚretrieved_modulesr^  rU  Ú_prefixs
   &&&&      r’   Úretrieve_modules_from_namesÚ+PreTrainedModel.retrieve_modules_from_namesž  s&  € Ù@EÓFÁ¸�s—x‘x §	¡	¨#£¨s°Ð 3Ö4ÁˆÐFð "×'Ñ'Ù6;Ób±e¨s¼sÀ3»xÈ!¹|Ô*ÐPSÐTVÕPW×P_ÑP_×PaÔ*ˆS�X‰X�c—i‘i “n S bÐ)Ö*±eÑbó
ˆð Ðà ×.Ñ.Ö0‰LˆDßØ!×3Ñ3Ð4°AÐ6�Ø×(Ñ(¨Ó1‘ßÜCFÀtÃ9ÈqÄ=�s—x‘x ×!7Ñ!7¸Ð >Ô?ÐVZ×VlÑVl�àÖ"Ø!×(Ñ(¨Ö0ñ 1ð !Ð ùò) Gùò
 cs   …,EÁEÁEÁ8'Ec                ó¦   € \        V\        4      '       g   VP                  p^ RIHu Hp \        W!4      '       g   \        V R24      hWn        R# )a%  
Register this class with a given auto class. This should only be used for custom models as the ones in the
library are already mapped with an auto class.



Args:
    auto_class (`str` or `type`, *optional*, defaults to `"AutoModel"`):
        The auto class to register this new model with.
Nz is not a valid auto class.)	r‡  r­   r²   Útransformers.models.autoÚmodelsrÞ  rÀ   rÛ   rÜ  )rg  Ú
auto_classÚauto_modules   && r’   Úregister_for_auto_classÚ'PreTrainedModel.register_for_auto_classµ  sE   € ô ˜*¤c×*Ò*Ø#×,Ñ,ˆJç6Ð6ä�{×/Ò/Ü 
˜|Ð+FÐGÓHÐHà$Žr•   c           
     ó  € \        V4      '       d   R# Vf   V P                  P                  f   R# V P                  P                  VRR
^ .3,          9   Ed/   Rp\        V P                  RR4      pV P                  P                  e0   V P                  P                  V P                  P                  8X  gf   V P                  P
                  e0   V P                  P
                  V P                  P                  8X  g   Vem   W@P                  P                  8X  dS   VRV P                  P                   RV P                  P                   RV P                  P
                   RV R	2	,          p\        P                  V4       R# R# )zf
Shows a one-time warning if the input_ids appear to contain padding and no attention mask was given.
Nrÿ  zÈWe strongly recommend passing in an `attention_mask` since your input_ids may be padded. See https://huggingface.co/docs/transformers/troubleshooting#incorrect-output-when-padding-tokens-arent-masked.Úsep_token_idz5
You may ignore this warning if your `pad_token_id` (z&) is identical to the `bos_token_id` (z), `eos_token_id` (z), or the `sep_token_id` (z ), and your input is not padded.rM  )rv   rÜ  Úpad_token_idr[  Úbos_token_idÚeos_token_idrÎ  rà  )rš   rT  r  Úwarn_stringr¬  s   &&&  r’   Ú%warn_if_padding_and_no_attention_maskÚ5PreTrainedModel.warn_if_padding_and_no_attention_maskË  sK  € ô �i× Ò ÙàÒ&¨D¯K©K×,DÑ,DÒ,LÙð �;‰;×#Ñ# y°°R¸°G°Õ'<Õ<ðFð ô # 4§;¡;°ÀÓEˆLà—‘×)Ñ)Ò5¸$¿+¹+×:RÑ:RÐVZ×VaÑVa×VnÑVnÔ:nØ—K‘K×,Ñ,Ò8¸T¿[¹[×=UÑ=UÐY]×YdÑYd×YqÑYqÔ=qØ Ò,°ÇÁ×AYÑAYÔ1YàØLÈTÏ[É[×MeÑMeÐLfð g.Ø.2¯k©k×.FÑ.FÐ-GÐGZÐ[_×[fÑ[f×[sÑ[sÐZtð u.Ø.:¨^Ð;[ð]õ�ô ×Ñ Ö,ñ- =r•   c                ó¦   € V P                   '       d   R# V P                  P                   '       d   R# V P                  P                  '       d   R# R# )z:
Returns whether the model has a tensor parallelism plan.
TF)rŒ  r@  rÜ  r˜  r™   s   &r’   Úsupports_tp_planÚ PreTrainedModel.supports_tp_planð  s9   € ð �=�=ˆ=Ùà�?‰?×#×#Ð#Ùà�;‰;×)×)Ð)ÙÙr•   c                ó   € V P                   # )z0
Returns the model's tensor parallelism degree.
)rÒ  r™   s   &r’   rQ  ÚPreTrainedModel.tp_size   s   € ð �}‰}Ðr•   c                ó¦   € V P                   '       d   R # V P                  P                   '       d   R # V P                  P                  '       d   R # R# )TF)r�  r@  rÜ  r–  r™   s   &r’   Úsupports_pp_planÚ PreTrainedModel.supports_pp_plan  s9   € ð �=�=ˆ=Ùà�?‰?×#×#Ð#Ùà�;‰;×)×)Ð)ÙÙr•   c                óÆ   € \        V R 4      '       d   V P                  # \        V RR4      pVe   V\        9  d   \        P                  RV R24       Rp\        V,          # )Ú_loss_functionr„  Nz`loss_type=zZ` was set in the config but it is unrecognized. Using the default loss: `ForCausalLMLoss`.ÚForCausalLM)rÀ   r¼  r[  rI   rÎ  rà  )rš   r„  s   & r’   Úloss_functionÚPreTrainedModel.loss_function  sh   € ä�4Ð)×*Ò*Ø×&Ñ&Ð&ä˜D +¨tÓ4ˆ	àÒ 	´Ô =Ü×ÑØ˜i˜[ð )=ð >ôð &ˆIÜ˜IÕ&Ð&r•   c                ó   € Wn         R # r—   )r¼  ©rš   rG  s   &&r’   r¾  r¿  $  s   € à#Ör•   c           	     ó´  € \        4       '       g   \        R4      h^ RIHpHpHp R pR p V P                  V4       V P                  '       g   VP                  MVf   VP                  MTpV P                  ew   ^ RIHp V P                  P                  '       * pV! V P                  P                  VR7      ;_uu_ 4        V! W! V P                  P                  R7      VR	7       RRR4       M%V! W! V P                  P                  R7      VR	7       R
V n        V P                  V4       R#   + '       g   i     L*; i  T P                  T4       i ; i)zYTemporarily register hidden kernel wrappers so `kernelize` can discover and replace them.z`Kernels are not available. To use kernels, please install kernels using `pip install -U kernels`)ÚDeviceÚModeÚ	kernelizec                 ó  € \        V R / 4      P                  4        Fe  w  rV\        V P                  4       4      9  g   K%  \	        V\
        P                  4      '       g   \        RV R24      hV P                  W4       Kg  	  R# )Ú_hidden_kernelsz#Attempted to register a kernel for zè, but it was not a `torch.nn.Module`. This means the underlying function needs to be decorated with `@use_kernel_func_from_hub`. Please submit and issue to the transformers repo: `https://github.com/huggingface/transformers/issues`.N)	r[  r9  r®   r›  r‡  r   rV  rÛ   Úregister_module)rU  r^  r^  s   &  r’   Úattach_hidden_kernelsÚ8PreTrainedModel.kernelize.<locals>.attach_hidden_kernels0  sz   € Ü# FÐ,=¸rÓB×HÑHÖJ‘�Øœt F×$9Ñ$9Ó$;Ó<Ö<Ü% b¬"¯)©)×4Ò4Ü(ØAÀ$Àð HFð Fóð ð
 ×*Ñ*¨4Ö4ó Kr•   c                 ój   € \        V R / 4       F!  p\        W4      '       g   K  \        W4       K#  	  R# )rÇ  N)r[  rÀ   rÌ  )rU  r^  s   & r’   Údetach_hidden_kernelsÚ8PreTrainedModel.kernelize.<locals>.detach_hidden_kernels;  s+   € Ü Ð(9¸2Ö>�ô ˜6×(Ô(Ü˜FÖ)ó	 ?r•   N)Úuse_kernel_mapping)Úinherit_mapping)r‰  )rä   ÚmodeT)rf   rÛ   ÚkernelsrÃ  rÄ  rÅ  r¡  ÚtrainingÚ	INFERENCEÚTRAININGrw  rÎ  Úuse_local_kernelÚkernel_mappingrä   r‰  Ú_use_kernels)	rš   rÐ  rÃ  rÄ  rÅ  rÉ  rÌ  rÎ  rÏ  s	   &&       r’   rÅ  ÚPreTrainedModel.kernelize(  s  € ä#×%Ò%ÜØróð ÷ 	4Ñ3ò		5ò	*ð	.Ø�J‰JÐ,Ô-à)-¯¯¨�4—>’>ÈTÊ\¸D¿MºMÐ_cˆDØ×!Ñ!Ò-Ý6à&*×&8Ñ&8×&IÑ&IÔ"I�Ù'¨×(:Ñ(:×(IÑ(IÐ[j×kÖkÙ˜d¨6°t·{±{×7GÑ7GÔ+HÈtÕT÷ lÐkñ ˜$ v°4·;±;×3CÑ3CÔ'DÈ4ÕPØ $ˆDÔð �J‰JÐ,Ö-÷ l×kûð �J‰JÐ,Õ-ús*   ­"E ÁA3E Ã&D1Ã)5E Ä1E	Ä<E ÅEc                ó    <€ V ^8„  d   QhRS[ /# r‹   rŽ   )r�   r‘   s   "€r’   r“   rW  T  s   ø€ ÷ 4ñ 4™Tñ 4r•   c                ó   € \        V R R4      # )r×  Frî  r™   s   &r’   rD  ÚPreTrainedModel.use_kernelsS  s   € ä�t˜^¨UÓ3Ð3r•   c                ó$   <€ V ^8„  d   QhRS[ RR/# )rŒ   rG  r�   NrŽ   )r�   r‘   s   "€r’   r“   rW  X  s   ø€ ÷ &ñ &¡ð &¨$ñ &r•   c                óä   € \        V4      '       d   \        V R R4      '       d   R# V'       d   V P                  4        R# \        V R R4      '       d   \        P	                  R4       RV n        R# )r×  FNzmDisabling kernels at runtime is a no-op as there is no 'unkernelize' routine; keeping current kernels active.)r�   r[  rÅ  rÎ  rà  r×  rÁ  s   &&r’   rD  rÛ  W  sX   € ô ��;Š;œ7 4¨¸×?Ò?ÙçØ�N‰NÖä�t˜^¨U×3Ò3Ü×#Ñ#ð Dôð !&ˆDÖr•   c                ó    <€ V ^8„  d   QhRS[ /# r‹   )r(   )r�   r‘   s   "€r’   r“   rW  f  s   ø€ ÷ 	ñ 	©ñ 	r•   c                óh   € V P                   P                  R8X  d   \        RRRR7      # \        4       # )a5  Build the default `CompileConfig` for `get_compiled_call`.

Inductor + `reduce-overhead` (the `CompileConfig` defaults) target CUDA.
torch_tpu registers its own TorchDynamo backend named `"tpu"`; route
`device.type == "tpu"` through it with static shapes to match the
common StaticCache + fixed-prefill usage.ÚtpuFrÌ  )ÚbackendÚdynamicrÐ  )rä   r‰  r(   r™   s   &r’   Ú_default_compile_configÚ'PreTrainedModel._default_compile_configf  s-   € ð �;‰;×Ñ˜uÔ$Ü ¨¸ÀIÔNÐNÜ‹Ðr•   c                ó4   <€ V ^8„  d   QhRS[ R,          RS[/# )rŒ   Úcompile_configNr�   )r(   r   )r�   r‘   s   "€r’   r“   rW  q  s    ø€ ÷ #ñ #±ÀÕ0Dð #Éñ #r•   c                ó¶  € RV P                   P                  9   d   V P                  # T;'       g    V P                  4       p\	        V P
                  RR4      ;'       g    V P                  4       p\        V R4      '       d   \	        V RV4      V8w  d;   Wn        \        P                  ! V P                  3/ VP                  4       B V n        V P                  # )at  Return a `torch.compile`'d version of `self.__call__`. This is useful to dynamically choose between
non-compiled/compiled `forward` during inference, especially to switch between prefill (where we don't
want to use compiled version to avoid recomputing the graph with new shapes) and iterative decoding
(where we want the speed-ups of compiled version with static shapes).Úllama4ræ  NÚ_compiled_callÚ_last_compile_config)rÜ  Ú
model_typeÚ__call__rã  r[  r‚  rÀ   rê  r¯   r  rr  ré  )rš   ræ  Údefault_configs   && r’   Úget_compiled_callÚ!PreTrainedModel.get_compiled_callq  s²   € ð �t—{‘{×-Ñ-Ô-Ø—=‘=Ð Ø'×IÐI¨4×+GÑ+GÓ+IˆÜ  ×!7Ñ!7Ð9IÈ4ÓP×rÐrÐTX×TpÑTpÓTrˆä˜Ð.×/Ò/Ü�tÐ3°^ÓDÈÔVà(6Ô%Ü"'§-¢-°·±Ñ"ZÀ×AWÑAWÓAYÑ"ZˆDÔØ×"Ñ"Ð"r•   c                ó   € V P                   # r—   )Ú_supports_attention_backend©rg  s   &r’   Úis_backend_compatibleÚ%PreTrainedModel.is_backend_compatibleƒ  s   € à×.Ñ.Ð.r•   c          
      ó`   <€ V ^8„  d   QhRS[ S[,          RS[R,          RRRS[R,          RR/# )rŒ   r   r£   Nr©   r¨   r˜   r�   )r°   r­   r®   rR   )r�   r‘   s   "€r’   r“   rW  ‡  sJ   ø€ ÷ /9ñ /9á™3•ið/9ñ ˜4•Kð/9ð -ð	/9ñ
 " DÕ(ð/9ð 
ñ/9r•   c                ó4  € VRJp\        4       '       d   V'       g   R# \        4       '       d•   \        4       '       g…   V'       g}   V P                  4        F)  w  rg\        P
                  ! VRR7      p\        WV4       K+  	  V P                  4        F)  w  ri\        P
                  ! V	RR7      p\        WV4       K+  	  R# WP                  P                  4       ,
           Fh  pV P                  V4      p\        W&RR7      p
\        P                  ! WzR7      pVe!   \        WWvRRVP                  4       V4       K\  \        WV4       Kj  	  V P                  4        F5  w  ri\        W&RR7      p\        P                  ! W›R7      p\        WV4       K7  	  R# )a³  Move the missing keys (keys that are part of the model parameters, but were NOT found in the loaded state dicts)
back from meta device to their device according to the `device_map` if any, else cpu. Takes care of sharding those
missing parameters if `device_mesh` is provided, i.e. we are using TP.
All non-persistent buffers are also moved back to the correct device (they are not part of the state_dict, but are
not missing either).
Nrâ   rí  T)Úvalid_torch_deviceF)r-   r.   rÎ   r(  r¯   Ú
zeros_liker©  r  rŽ  r;  r%  r4   Ú
empty_likerG   Úget_local_rankÚnamed_non_persistent_buffers)rš   r   r£   r©   r˜   r›   r  rï  rG  rò  Úparam_deviceÚbuffer_devices   &&&&&       r’   r•  Ú6PreTrainedModel._move_missing_keys_from_meta_to_device‡  sW  € ð $¨4Ð/ˆä%×'Ò'·Ùô ×ÒÔ%9×%;Ò%;ÇLØ"×3Ñ3Ö5‘
�Ü×(Ò(¨°uÔ=�Ü*¨4°eÖ<ñ 6ð  $×1Ñ1Ö3‘�Ü×(Ò(¨¸Ô>�Ü*¨4°eÖ<ñ  4ñ ð
  ×"<Ñ"<×"AÑ"AÓ"C×CÐCˆCØ×0Ñ0°Ó5ˆEÜ% jÈ$ÔOˆLÜ×$Ò$ UÔ@ˆEàÒ&Ü+Ø ¨T°5¸+×:TÑ:TÓ:VÐXcöô
 +¨4°eÖ<ñ Dð  ×<Ñ<Ö>‰KˆCÜ& zÈ4ÔPˆMÜ×$Ò$ VÔBˆEÜ& t°%Ö8ó ?r•   c                ó$   <€ V ^8„  d   QhRS[ RR/# )rŒ   r›   r�   NrŽ   )r�   r‘   s   "€r’   r“   rW  ¸  s   ø€ ÷ "&ñ "&±Tð "&¸dñ "&r•   c           
     ó~  € \        4       '       dH   \        4       '       g8   V P                  4        F  p V P                  V4      pRVn        K  	  RV n        \        4       '       d›   V'       g“   ^ RIp\        V P                  RR7      P                  4        Uu0 uF  p\        VRR4      '       d   K  VkK  	  up4      pVP                  P                  V^ R7      ;_uu_ 4        V P                  4        RRR4       R# V P                  4        R#   \
         d     Kñ  i ; iu upi   + '       g   i     R# ; i)a  
Initialize the missing keys (keys that are part of the model parameters, but were NOT found in the loaded state dicts), according to
`_initialize_weights`. Indeed, since the corresponding weights are missing from the state dict, they will not be replaced and need to
be initialized correctly (i.e. weight initialization distribution).

Also marks non-missing params/buffers with `_is_hf_initialized` and propagates this flag to modules,
so that `_initialize_weights` can skip fully-initialized modules entirely.
TN)Ú	keep_varsrÆ  FrD  )r.   rÎ   rñ   r%  rÆ  ÚAttributeErrorr-   rÝ  r°   rì   r[  rà  rG  r   )rš   r›   r  Úparam_or_bufferrÝ  rE  Únot_initialized_parameterss   &&     r’   r—  Ú(PreTrainedModel._initialize_missing_keys¸  s	  € ô ×ÒÔ%9×%;Ò%;ð —‘Ö(�ðØ&*×&BÑ&BÀ3Ó&G�OØ9=�OÖ6ñ )ð '+ˆDÔ#ô &×'Ò'·Ûô *.Ø ŸO™O°d˜OÓ;×BÑBÔDÓtÑD�qÌGÐTUÐWkÐmr×Ls—�ÑDÑtó*Ð&ð —‘×2Ñ2Ð3MÐ]^Ð2×_Õ_Ø×'Ñ'Ô)÷ `Ñ_ð ×#Ñ#Ö%øô &ô Úðüò uç_×_Ð_ús)   µDÂD&Â5D&Ã'D+ÄD#Ä"D#Ä+D<	c                ó$   <€ V ^8„  d   QhRS[ RR/# )rŒ   r|  r�   N)rw   )r�   r‘   s   "€r’   r“   rW  Ü  s   ø€ ÷  ñ  Ñ@Qð  ÐVZñ  r•   c                ó  € \         ;QJ d*    R V P                  4        4       F  '       g   K   RM	  RM! R V P                  4        4       4      pV'       d   R0M	\        4       p\         ;QJ d*    R V P                  4        4       F  '       g   K   RM	  RM! R V P                  4        4       4      pV'       d   VP                  R4       V P                  ;'       g    \        4       pV P
                  ;'       g    \        4       V,          pRRr‡\        V4      ^ 8”  d-   \        P                  ! RP                  R	 V 4       4      4      p\        V4      ^ 8”  d-   \        P                  ! RP                  R
 V 4       4      4      pVe6   VP                   U	u0 uF  q—P                  V	4      e   K  V	kK  	  up	Vn
        Ve8   VP                   U	u0 uF  q˜P                  V	4      e   K  V	kK  	  up	Vn        R# R# u up	i u up	i )z¡Adjust the `missing_keys` and `unexpected_keys` based on current model's exception rules, to avoid
raising unneeded warnings/errors. This is performed in-place.
c              3   óH   "  € T F  w  rVP                  R 4      x € K  	  R# 5i)zrotary_emb.inv_freqN©r6  ©r  rò  rp  s   &  r’   r  ÚFPreTrainedModel._adjust_missing_and_unexpected_keys.<locals>.<genexpr>ã  s!   é € Ð"pÑ[oÉiÈf 6§?¡?Ð3H×#IÐ#IÓ[oùó   ‚ "TFzrotary_emb\.inv_freqc              3   óH   "  € T F  w  rVP                  R 4      x € K  	  R# 5i)Úposition_idsNr	  r
  s   &  r’   r  r  æ  s    é € Ð&mÑXlÉ9È6 v§¡°~×'FÐ'FÓXlùr  z(^|\.)position_ids$Nrs  c              3   ó.   "  € T F  pR V R2x € K  	  R# 5i©rr  r  Nr±   ©r  Úpatterns   & r’   r  r  î  s   é € Ð6gÑVfÈ7¸!¸G¸9ÀAµÓVfùr   c              3   ó.   "  € T F  pR V R2x € K  	  R# 5ir  r±   r  s   & r’   r  r  ð  s   é € Ð9mÑYlÈg¸Q¸w¸iÀq½/ÓYlùr   )r‹  r  rf  rk  r”  r“  rí   rƒ  r  rË  r   r„  rƒ  )
rš   r|  Úhas_inv_freq_buffersÚadditional_unexpected_patternsÚhas_position_ids_buffersÚmissing_patternsÚunexpected_patternsÚignore_missing_regexÚignore_unexpected_regexr  s
   &&        r’   r˜  Ú3PreTrainedModel._adjust_missing_and_unexpected_keysÜ  s§  € ÷  #›sÑ"pÐ[_×[mÑ[mÔ[oÓ"pŸsŸsšsÑ"pÐ[_×[mÑ[mÔ[oÓ"pÓpÐßFZÐ*AÑ)BÔ`cÓ`eÐ&ç#&£3Ñ&mÐX\×XjÑXjÔXlÓ&m§3§3¢3Ñ&mÐX\×XjÑXjÔXlÓ&mÓ#mÐ ß#Ø*×.Ñ.Ð/EÔFà×?Ñ?×HÐHÄ3Ã5ÐØ#×FÑF×OÐOÌ#Ë%ÐSqÕqÐØ8<¸dÐ5ÜÐÓ  1Ô$Ü#%§:¢:¨c¯h©hÑ6gÑVfÓ6gÓ.gÓ#hÐ ÜÐ"Ó# aÔ'Ü&(§j¢j°·±Ñ9mÑYlÓ9mÓ1mÓ&nÐ#ð  Ò+à+×8Ò8ó)Ù8˜×<WÑ<WÐX[Ó<\—�Ñ8ñ)ˆLÔ%ð
 #Ò.à+×;Ò;ó,Ù;˜×?]Ñ?]Ð^aÓ?b—�Ñ;ñ,ˆLÖ(ñ /ùò)ùò,s   ÆHÆ8HÇHÇ1Hc                ój  € \        V R/ 4      P                  4        F!  pV P                  V4      p\        VRR4       K#  	  V P	                  4       '       dX   VP
                   Uu0 uF9  pW@P                  9   g%   \        V P                  V4      RR4      '       d   K7  VkK;  	  upVn        R# R# u upi )a-  Adds the `_is_hf_initialized` flag on parameters that will be tied, in order to avoid initializing them
later as they will be tied (overwritten) anyway.
This is very important as most embeddings are tied, and they are huge params (vocabularies are often 256k), so
running inits on them is very costly.rŽ  rÆ  TFN)r[  r;  rŠ  r¦  r´  r   rŽ  r%  )rš   r|  Ú
tied_paramrï  r  s   &&   r’   r”  Ú0PreTrainedModel.mark_tied_weights_as_initializedþ  s©   € ô
 " $Ð(?ÀÓD×IÑIÖKˆJØ×&Ñ& zÓ2ˆEÜ�EÐ/°Ö6ñ Lð ×Ñ× Ò ð
 (×4Ò4ó)á4�CØ×4Ñ4Ô4Ü˜t×;Ñ;¸CÓ@ÐBVÐX]×^÷ �Ù4ñ)ˆLÖ%ñ !ùò)s   Á%4B0ÂB0c                ó    <€ V ^8„  d   QhRS[ /# )rŒ   rÇ  r­  )r�   r‘   s   "€r’   r“   rW    s   ø€ ÷ ]ñ ]©cñ ]r•   c                ó²  €  V P                  V4      #   \         d     Mi ; i T P                  T4      #   \         d     Mi ; i\        Y4      w  r#TR8X  dp   \	        TP
                  R\        P                  P                  P                  4      \        P                  P                  P                  Jd   TP                  4       # \        RT R24      h)aI  
Return the parameter or buffer given by `target` if it exists, otherwise throw an error. This combines
`get_parameter()` and `get_buffer()` in a single handy function. If the target is an `_extra_state` attribute,
it will return the extra state provided by the module. Note that it only work if `target` is a leaf of the model.
Ú_extra_stateÚget_extra_stateÚ`z2` is neither a parameter, buffer, nor extra state.)
rŠ  r  Ú
get_bufferrT   r[  rC  r¯   r   rV  r"  )rš   rÇ  rU  r¡  s   &&  r’   r%  Ú'PreTrainedModel.get_parameter_or_buffer  sÂ   € ð	Ø×%Ñ% fÓ-Ð-øÜô 	Ùð	úð	Ø—?‘? 6Ó*Ð*øÜô 	Ùð	úä1°$Ó?Ñˆà˜.Ô(Ü˜×(Ñ(Ð*;¼U¿X¹X¿_¹_×=\Ñ=\Ó]Ü—8‘8—?‘?×2Ñ2ó3ð ×)Ñ)Ó+Ð+ä˜q  Ð(ZÐ[Ó\Ð\s   ‚ “! !¥6 ¶AÁAc          	      óf   <€ V ^8„  d   QhRS[ RS[ RS[S[S[S[P
                  3,          ,          /# )rŒ   rð  r  r�   )r�   r   rg  r­   r¯   r   )r�   r‘   s   "€r’   r“   rW  2  s8   ø€ ÷ #ñ #Ùð#Ù6:ð#á	‘%™™UŸ\™\Ð)Õ*Õ	+ñ#r•   c              #  óÒ   "  € V P                  WR7       FL  w  r4RV9   d   VP                  R^4      MRV3w  rVV P                  V4      pWeP                  9   g   KG  W43x € KN  	  R# 5i)z—Similar to `named_buffers`, but only yield non-persistent ones. It is handy as it's not perfectly straightforward
to know if they are persistent or not)rð  r  rY  rÀ  N)r  r¯  r&  Ú_non_persistent_buffers_set)rš   rð  r  r^  rã   r§  Úbuf_names   &&&    r’   rû  Ú,PreTrainedModel.named_non_persistent_buffers2  sg   é € ð
 !×.Ñ.°wÐ.Öb‰LˆDð 7:¸T´k˜tŸ{™{¨3°Ô2ÈÈDÀzÑˆFØ×'Ñ'¨Ó/ˆFØ×=Ñ=Ö=Ø�lÔ"ó cùs   ‚AA'ÁA'c                ó    <€ V ^8„  d   QhRS[ /# )rŒ   rÐ  rŽ   )r�   r‘   s   "€r’   r“   rW  ?  s   ø€ ÷ ñ ™$ñ r•   c                ój   <€ \         SV `  V4      pV P                  '       d   V P                  4        V# r—   )rb  ÚtrainrD  rÅ  )rš   rÐ  ÚoutrC  s   && €r’   r-  ÚPreTrainedModel.train?  s,   ø€ Ü‰g‰m˜DÓ!ˆØ××ÐØ�N‰NÔØˆ
r•   c                ó$   € V P                  R 4      # ©F)r-  r™   s   &r’   ro  ÚPreTrainedModel.evalE  s   € Ø�z‰z˜%Ó Ð r•   c                ó    <€ V ^8„  d   QhRS[ /# r‹   rŽ   )r�   r‘   s   "€r’   r“   rW  I  s   ø€ ÷ +ñ +™tñ +r•   c                ó   € V P                   R J# r—   )rÜ  rò  s   &r’   r´  ÚPreTrainedModel.is_remote_codeH  s   € à�‰ dÐ*Ð*r•   c                óè  <€ V ^8„  d   Qh/ S[ S[,          R,          ;R&   S[ S[,          ;R&   S[;R&   S[;R&   S[S[,          R,          ;R&   S[;R&   S[S[S[,          ,          ;R&   S[S[,          S[S[,          ,          R,          ;R	&   S[S[,          S[S[,          ,          R,          ;R
&   S[S[,          S[S[,          ,          R,          ;R&   S[S[,          S[S[,          ,          R,          ;R&   S[S[S[3,          ;R&   S[S[,          S[S[,          ,          R,          ;R&   S[S[,          S[S[,          ,          R,          ;R&   S[S[,          S[S[,          ,          R,          ;R&   S[;R&   S[;R&   S[;R&   S[S[,          R,          ;R&   S[S[S[3,          ;R&   S[S[S[S[S[3,          3,          ;R&   S[;R&   S[;R&   S[;R&   S[R,          ;R&   # )rŒ   Nra  r€  rê  Ú_is_statefulrÑ  r¢  Úinput_modalitiesr‘  r’  r�  r�  rX  r”  r“  r•  r  r  r-  r3  rŒ  r�  rÊ  Ú_can_compile_fullgraphrñ  rY  )	r‰  r    r)   r­   r�   r°   rf  r®   rg  )r�   r‘   s   "€r’   r“   rW  ž  s?  ø‡ ‚ ñ* Ñ'Õ(¨4Õ/Ñ6ñ+ ñ, "Ñ"2Õ3ÑFñ- ñ0 Ññ1 ñ2 Ññ3 ñ4 ‘S•	˜DÕ Ñ'ñ5 ñ: Ñ&ñ; ñ@ ™D¡�I•oÑ.ñA ñF ™3•x¡$¡s¥)Õ+¨dÕ2Ñ9ñG ñH "%¡S¥©D±­IÕ!5¸Õ!<ÑCñI ñR ™s�8¡d©3¥iÕ/°$Õ6Ñ=ñS ñT #&¡c¥(©T±#­YÕ"6¸Õ"=ÑDñU ñ\ ™S¡#˜X�Ñ-ñ] ñ` &)©¥X±±Sµ	Õ%9¸DÕ%@ÑGña ñd ),©C­±4¹µ9Õ(<¸tÕ(CÑJñe ñh !¡�X©©S­	Õ1°DÕ8Ñ?ñi ñn Ñ ño ñp Ñ&ñq ñr Ñ%ñs ñv (,©C¥y°4Õ'7Ñ>ñw ñB ‘3™�8�nÑ#ñC ñN ‘3™™c¡3˜h�Ð'Õ(Ñ/ñO ñT &*Ñ1ñU ñV !Ñ(ñW ñ^ "&Ñ-ñ_ ñb  �Ñ+òc r•   )ré  r‹  rª  rÆ  r�  r�  r”  r“  r•  rê  r¼  r‘  r�  rŠ  r‰  r’  rŒ  r×  rŽ  rÜ  r‚  rÉ  rw  r„  r~  rÑ  rv  rD  rH  r—   )r±   Nr1  r3  )NT)NNT)NFT)TNFÚ50GBNNTT)T)Ú	AutoModel)TT)‹r²   r³   r´   rµ   r¶   ra  r)   r€  rÜ  rê  r7  rÑ  r¢  r8  r‘  r’  r�  r�  rX  r”  r“  r•  r  r  r-  r3  rŒ  rÒ  r�  rÊ  r9  rñ  rY  r·   r¯   ÚcompilerÚallow_in_graphrZ  r^  rc  rt  r¨  r¯  r´  ÚsetterrÆ  rž  rÓ  Úclassmethodræ  r@  r  r
  r  r$  r)  r.  rx  r|  r9  rD  rl  r(  rx  r  r�  r•  rŸ  r²  rµ  r»  Úno_gradré  rõ  rÞ  Úguard_torch_init_functionsr   rš  rã  r'  rJ  rF  rN  rO  rY  rc  rd  re  r�  r“  r�  rË  r   r   r®  rµ  rÝ  r   r^   rº  r  r   rV  r  r:  r-  r1  r7  r<  rE  rh  r4  rm  rn  r¢  r©  r±  r´  rQ  r¹  r¾  rÅ  rD  rã  rî  ró  r•  r—  r˜  r”  r%  rû  r-  ro  r´  r¸   r¹   rº   Ú__classcell__©rC  r‘   s   @@r’   r„   r„   ž  sY  ù‡ € ñð( 37€LØ6FÐØ€KØÐØ€LØ#'€Jð '€Oð )/Ðð 6:ÐØ?CÐð
 :>ÐØ@DÐ ð *.ÐàCGÐ#àFJÐ&à;?Ðð !€NØ!&ÐØ %Ðà:>Ð%ð  $€Hà€Hð ,0€Hð -2Ð#Ø#(Ðð ).Ðà'+ÐàØ
‡^�^×"Ñ"÷'.ó #ó ð'.ðR ÷9ó ð9õ/÷0-Mó -Mò^?>ðB ÷ó ðð ÷ó ðð ‡^�^÷!ó ð!ðF ‡^�^÷ó ðô
:ò;÷,ð ,ð@ ñGó ðGðR ÷;ó ð;ð ÷$ó ð$÷L@ò @÷DIò I÷Vò ÷<	ð 	÷ò ÷6q.ò q.÷f1ð 1÷$$ò $$÷L"ð "ð0 ÷ó ðð$ ÷5ó ð5÷d8ò d8÷L6Wð 6Wòp*òX)÷ò ÷@%ò %ò4ò.%ð" ‡]‚]ƒ_ñ=?ó ð=?÷~)ò )ð2 ‡]‚]ƒ_Ø	×$Ò$Ó&ñHó 'ó ðH÷8p%ò p%÷dQ8ò Q8òf
M÷9ò 9ôv%+÷N]ò ]÷~Aò AòF÷2@ò @ò,dò
d÷
ð 
÷
ð 
ò
2ô(.ðT :>Ðgq÷ ò ò,/ð( ÷nó ðn÷\ò \ñ| ˆ>×%Ñ%Ó&ô4ó 'ð4ôñ$ ˆ5�8‰8�?‰?×ÑÓ ô-ó !ð-ñ2 ˆ5�8‰8�?‰?×ÑÓô:+ó ð:+õx'õ(ð ÷ó ð÷>ð ÷ "&ò "&ðH ðSð ?Cð	Sð
 /3ðSð ).ðSð  %ðSð "'ðSð $(ðSð ðSð (,ðSð "ðSð BFðSð %)÷Sñ Só ðSðj ÷c0ñ c0ó ðc0ðJ ÷$ó ð$ôL!ð. ó%ó ð%ò*#-ðJ ñó ðð ñó ðð ñ
ó ð
ð ñ'ó ð'ð ×Ññ$ó ð$ô).ðV ÷4ó ð4ð ×Ñ÷&ó ð&÷	ð 	÷#ð #ð$ ñ/ó ð/÷/9ð /9÷b"&ð "&÷H ð  òD÷8]ð ]÷0#ò #÷õ ò!ð ÷+ó ð+÷Wu … r•   r~  r;  z
model file)ÚobjectÚobject_classÚobject_filesc                ó<   € V ^8„  d   QhR\         R\        R\         /# ©rŒ   r~  Ú	recursiver�   )r„   r�   )r�   s   "r’   r“   r“   U  s   € × YÑ YœÐ Y´DÐ YÄ_Ñ Yr•   c                 ó   € R # r—   r±   ©r~  rI  s   &&r’   rÙ  rÙ  T  s   € ÙVYr•   c                ód   € V ^8„  d   QhR\         P                  R\        R\         P                  /# rH  ©r   rV  r�   )r�   s   "r’   r“   r“   Y  s   € × MÑ MœŸ	™	Ð M¬dÐ M¼r¿y¹yÑ Mr•   c                 ó   € R # r—   r±   rK  s   &&r’   rÙ  rÙ  X  s   € ÙJMr•   c                ód   € V ^8„  d   QhR\         P                  R\        R\         P                  /# rH  rM  )r�   s   "r’   r“   r“   \  s)   € ÷ ñ œŸ	™	ð ¬dð ¼r¿y¹yñ r•   c                ó¨   € \        4       '       d   / pV'       d   WR&   \        V 3/ VB # \        V R4      '       d   \        V P                  4      # V # )a…  
Recursively unwraps a model from potential containers (as used in distributed training).

Args:
    model (`torch.nn.Module`): The model to unwrap.
    recursive (`bool`, *optional*, defaults to `False`):
        Whether to recursively extract all cases of `module.module` from `model` as well as unwrap child sublayers
        recursively, not just the top-level distributed containers.
rI  rU  )rc   r}   rÀ   rÙ  rU  )r~  rI  rÉ  s   && r’   rÙ  rÙ  \  sP   € ô × Ò ØˆßØ"+�;ÑÜ*¨5Ñ;°FÑ;Ð;ô �5˜(×#Ò#Ü §¡Ó-Ð-àˆLr•   c                óp   € V ^8„  d   QhR\         \        ,          \        P                  ,          R\        /# )rŒ   rä   r�   )r­   rÅ   r¯   rä   r�   )r�   s   "r’   r“   r“   u  s+   € ÷ @ñ @¤#¬¥)¬e¯l©lÕ":ð @¼tñ @r•   c                óZ   € V R8X  d   R# \         P                  ! V 4      P                  R9  # )z‡Check if the device is an accelerator. We need to function, as device_map can be "disk" as well, which is not
a proper `torch.device`.
rÅ  F)r0  râ   )r¯   rä   r‰  rí  s   &r’   Úis_accelerator_devicerS  u  s)   € ð �ÔÙä�|Š|˜FÓ#×(Ñ(°Ñ?Ð?r•   c                óJ   € V ^8„  d   QhR\         R\        R\        R,          /# )rŒ   r~  Úaccelerator_device_mapr˜   N©r„   r®   rR   )r�   s   "r’   r“   r“     s*   € ÷ ñ ÜðÜ48ðÜHSÐVZÕHZñr•   c                ó  € \        R 4      pV P                  P                  4       p\        4       '       d   V P                  M. pVP                  4        F©  w  rgWd9   d   K  V P                  V4      pVe   VP                  WV4      p	MVP                  4       p	VP                  4       V	,          p
\        V4      ^ 8”  d*   \        WeRR7      RJpY«'       d   \        4       M^,          p
W7;;,          V
,          uu&   K«  	  V# )zÈ
This utility function calculates the total bytes count needed to load the model on each device.
This is useful for caching_allocator_warmup as we want to know how much cache we need to pre-allocate.
c                  ó   € ^ # )r   r±   r±   r•   r’   r  Ú&get_total_byte_count.<locals>.<lambda>‡  s   € ©1r•   NT)Ú	is_weight)r   rŽ  r;  rÂ   r¯  r9  r%  Úparam_element_sizerQ  r+  rí   rC   rÈ   )r~  rU  r˜   Útotal_byte_countÚtied_param_namesr¯  r¡  rä   rï  Ú
dtype_sizeÚparam_byte_countÚis_part_of_plans   &&&         r’   Úget_total_byte_countra    sç   € ô #¡9Ó-ÐØ×2Ñ2×7Ñ7Ó9ÐÜ@×BÒBˆe�mŠmÈ€Gà4×:Ñ:Ö<Ñˆ
àÔ)Ùà×-Ñ-¨jÓ9ˆàÒ#Ø%×8Ñ8¸ÈEÓR‰Jà×+Ñ+Ó-ˆJà Ÿ;™;›=¨:Õ5Ðäˆw‹<˜!ÔÜ4°ZÐTXÔYÐaeÐeˆOØÎÔ!BÔ!DÐ]^Õ^Ðà× Ð$4Õ4Õ ñ% =ð& Ðr•   c                óJ   € V ^8„  d   QhR\         R\        R\        R,          /# )rŒ   r~  r‹  r˜   NrV  )r�   s   "r’   r“   r“   ¡  s2   € ÷ Fgñ Fg¤Oð FgÌ$ð FgÔ^iÐlpÕ^pñ Fgr•   c                ód  € VP                  4        UUu/ uF/  w  r4\        V4      '       g   K  V\        P                  ! V4      bK1  	  pppV'       g   R# \	        WV4      pVP                  4        EF3  w  rGVP
                  R9   dÒ   \        \        VP
                  4      pVP                  e   VP                  MVP                  4       p	VP                  V	4      w  r«VP                  V	4      VP                  V	4      ,
          pW|,
          V8”  d
   W|,
          pM*W|,
          R8”  d   V^,           V
8”  d   ^ pMV^,           pM^ p\        W{R,
          4      pMVP
                  R8X  d   Kû  \        P                  ! \        V^,          4      \        P                  VRR7      pEK6  	  R# u uppi )a  This function warm-ups the caching allocator based on the size of the model tensors that will reside on each
device. It allows to have one large call to Malloc, instead of recursively calling it later when loading
the model, which is actually the loading speed bottleneck.
Calling this function allows to cut the model loading time by a very large margin.

A few facts related to loading speed (taking into account the use of this function):
- When loading a model the first time, it is usually slower than the subsequent times, because the OS is very likely
to cache the different state dicts (if enough resources/RAM are available)
- Trying to force the OS to cache the files in advance (by e.g. accessing a small portion of them) is really hard,
and not a good idea in general as this is low level OS optimizations that depend on resource usage anyway
- As of 18/03/2025, loading a Llama 70B model with TP takes ~1 min without file cache, and ~13s with full file cache.
The baseline, i.e. only loading the tensor shards on device and adjusting dtype (i.e. copying them) is ~5s with full cache.
These numbers are reported for TP on 4 H100 GPUs.
- It is useless to pre-allocate more than the model size in this function (i.e. using an `allocation_factor` > 1) as
cudaMalloc is not a bottleneck at all anymore
- Loading speed bottleneck is now almost only tensor copy (i.e. changing the dtype) and moving the tensors to the devices.
However, we cannot really improve on those aspects obviously, as the data needs to be moved/copied in the end.
NÚmpsF)r¦   rä   r£  )r  Úxpug      ØAg333333ÓA)r9  rS  r¯   rä   ra  r‰  r[  rþ  Úcurrent_deviceÚmem_get_infoÚmemory_reservedÚmemory_allocatedr  r?  rÅ   r  )r~  r‹  r˜   rï  rä   rU  r\  Ú
byte_countÚaccelerator_modulerþ  Úfree_device_memoryÚtotal_device_memoryÚunused_memoryrp  s   &&&           r’   rˆ  rˆ  ¡  sx  € ð* :M×9RÑ9RÔ9TôÙ9T©¨ÔXmÐnt×XuÔ#ˆŒu�|Š|˜FÓ#Ò#Ñ9Tð ñ ÷ "Ùä+¨EÈ<ÓXÐð /×4Ñ4×6ÑˆØ�;‰;˜/Ô)Ü!(¬°·±Ó!<ÐØ$*§L¡LÒ$<�F—L’LÐBT×BcÑBcÓBeˆEØ6H×6UÑ6UÐV[Ó6\Ñ3ÐØ.×>Ñ>¸uÓEÐHZ×HkÑHkÐlqÓHrÕrˆMð Õ)¨MÔ9Ø'Õ7‘
àÕ+¨mÔ;ð ! 1Õ$Ð'9Ô9Ø!"‘Jð "/°Õ!2‘Jð �
ô ˜Z¸}Õ)LÓM‰JØ�[‰[˜EÔ!ñ
 ä�KŠKœ˜J¨!�OÓ,´E·M±MÈ&Ð`eÔf‹óS 7ùós
   ”F,®F,c                   ón   a a€ ] tR tRt oRtR]R]R]R]R]R]R	]R
]R]	R]
/
tV3R lV 3R lltRtVtV ;t# )ÚAttentionInterfaceiê  aO  
Dict-like object keeping track of allowed attention functions. You can easily add a new attention function
with a call to `register()`. If a model needs to locally overwrite an existing attention function, say `sdpa`,
it needs to declare a new instance of this class inside the `modeling_<model>.py`, and declare it on that instance.
Úflash_attention_4Úflash_attention_3r7  rO  rJ  zpaged|flash_attention_4zpaged|flash_attention_3zpaged|flash_attention_2z
paged|sdpazpaged|eagerc                ó,   <€ V ^8„  d   QhRS[ RS[RS[/# )rŒ   r×  rÌ  r�   )r­   r   )r�   r‘   s   "€r’   r“   ÚAttentionInterface.__annotate__   s"   ø€ ÷ 9ñ 9±ð 9¹xð 9ÉHñ 9r•   c                óŽ   <€ Vf   \         P                  R4       MVR8w  d   W9  d   \        RV R24      h\        SV `  W4      # )zcReturn the requested `attn_implementation`. Also strictly check its validity, and raise if invalid.a	  You tried to access the `AttentionInterface` with a `config._attn_implementation` set to `None`. This is expected if you use an Attention Module as a standalone Module. If this is not the case, something went wrong with the dispatch of `config._attn_implementation`rK  r#  zP` is not a valid attention implementation registered in the `AttentionInterface`)rÎ  rà  ÚKeyErrorrb  rÍ   )rš   r×  rÌ  rC  s   &&&€r’   Úget_interfaceÚ AttentionInterface.get_interface   s[   ø€ àÒ&Ü×ÑðKõð
 ! GÔ+Ð0CÔ0OÜØÐ'Ð(Ð(xÐyóð ô ‰w‰{Ð.Ó8Ð8r•   r±   )r²   r³   r´   rµ   r¶   r9   r;   r@   r:   rA   r7   Ú_global_mappingrw  r¹   rº   rB  rC  s   @@r’   rp  rp  ê  s^   ù‡ € ñð 	Ð4ØÐ4ØÐ4ØÐ0ØÐ&Ø!Ð#:Ø!Ð#:Ø!Ð#:ØÐ2ØÐ4ð€O÷9÷ 9ó 9r•   rp  c                   ó\   a € ] tR tRt o Rt]V 3R lR l4       t]V 3R lR l4       tRtV t	R# )	ÚPreTrainedAudioTokenizerBasei  a�  
Class that additionally defines the behavior of any `audio_tokenizer` to be added.
Characteristic for any of them:
    1. Encode raw audio into discrete audio codebooks (with x channels)
    2. Decode from discrete audio codebooks back to raw audio
It is possible that they can decode in different ways given a different representation
but they are forced to support 2. nonetheless, e.g. see `DAC`.
c                ó4   <€ V ^8„  d   QhRS[ P                  /# )rŒ   Úinput_values©r¯   r   )r�   r‘   s   "€r’   r“   Ú)PreTrainedAudioTokenizerBase.__annotate__  s   ø€ ÷ ñ ¡5§<¡<ñ r•   c                ó   € R# )zq
Encode raw audio retrieved from a respective `FeatureExtractor` into discrete audio codebooks (with x channels)
Nr±   )rš   r}  rÈ  rÉ  s   &&*,r’   ÚencodeÚ#PreTrainedAudioTokenizerBase.encode  ó   ‚ r•   c                ó4   <€ V ^8„  d   QhRS[ P                  /# )rŒ   Úaudio_codesr~  )r�   r‘   s   "€r’   r“   r  $  s   ø€ ÷ Eñ E¡%§,¡,ñ Er•   c                ó   € R# )z6Decode from discrete audio codebooks back to raw audioNr±   )rš   r…  rÈ  rÉ  s   &&*,r’   ÚdecodeÚ#PreTrainedAudioTokenizerBase.decode#  rƒ  r•   r±   N)
r²   r³   r´   rµ   r¶   r   r�  r‡  r¹   rº   r»   s   @r’   r{  r{    s4   ø‡ € ñð ÷ó ðð
 ÷Eó öEr•   r{  c                ó@   € V ^8„  d   Qh/ ^ \         9   d
   \        ;R&   # )rŒ   rQ  )Ú__conditional_annotations__rp  )r�   s   "r’   r“   r“      s    € × Ñ ÷B` CÒ BÔ+Ñ BòC` r•   r—   )râ   TN)NNNr1  (  rŠ  ry  r—  rž  rd  rî  rË   rƒ  r  r¼  Úabcr   r   Úcollections.abcr   r   Ú
contextlibr   Údataclassesr   r	   r
   r   Ú	itertoolsr   Ú	threadingr   Útypingr   r   r   r   r   Úzipfiler   r¯   Úhuggingface_hubr   r   Ú	packagingr   Úsafetensorsr   Úsafetensors.torchr   r7  r   rí  r   r   Útorch.distributionsr   Útorch.utils.checkpointr   rÀ  r   rÞ  Úconfiguration_utilsr    Úconversion_mappingr!   Úcore_model_loadingr"   r#   r$   r%   rÁ   r&   Údynamic_module_utilsr'   Ú
generationr(   r)   Úintegrationsr*   r+   r,   r-   r.   Úintegrations.accelerater/   r0   r1   r2   r3   r4   r5   râ  r6   Úintegrations.eager_pagedr7   Úintegrations.finegrained_fp8r8   Úintegrations.flash_attentionr9   Úintegrations.flash_pagedr:   Úintegrations.flex_attentionr;   rA  r<   r=   Úintegrations.moer>   Úintegrations.peftr?   Úintegrations.sdpa_attentionr@   Úintegrations.sdpa_pagedrA   Úintegrations.tensor_parallelrB   rC   rD   rE   rF   rG   rH   Úloss.loss_utilsrI   Úmodeling_flash_attention_utilsrJ   rK   rL   rM   Úmodeling_rope_utilsrN   Úmonkey_patchingrO   rP   Úpytorch_utilsrQ   Ú
quantizersrR   Úquantizers.autorS   Úquantizers.quantizers_utilsrT   Úsafetensors_conversionrU   ÚutilsrV   rW   rX   rY   rZ   r[   r\   r]   r^   r_   r`   ra   rb   rc   rd   re   rf   rg   rh   ri   rj   Úutils.genericrk   rl   rm   Ú	utils.hubrn   ro   rp   rq   Úutils.import_utilsrr   rs   rt   ru   rv   Úutils.loading_reportrw   rx   Úutils.output_capturingry   rz   Úutils.quantization_configr{   Úaccelerate.hooksr|   Úaccelerate.utilsr}   Ú_typingr~   Úis_availabler¿   Ú!smdistributed.modelparallel.torchÚmodelparallelrâ  Úsmdistributed.modelparallelr   ÚSMP_VERSIONr   rá  Ú
get_loggerr²   rÎ  rÌ   rÍ   Úupperr€   r‚   rƒ   rÑ   rÕ   rˆ   rÂ   rÈ   rÎ   rÒ   rÖ   rà   rè   ró   r�   Úuint8Úint8Úint16Úuint16r  r  Úint32Úuint32rî   Úfloat64Úint64Úuint64Úfloat8_e4m3fnÚfloat8_e5m2r>  r)  rJ  rS  ra  rv  r|  rŸ  r©  r°  rÚ  rå  rç  r6  rV  r„   rº  r¶   r�   rÙ  rS  ra  rˆ  rp  rQ  r{  r“   )rŠ  s   @r’   Ú<module>rÏ     s?  øð÷ Ñ Û Û Û Û Û 	Û 	Û 
Û Ý Ý #ß .Ý %ß (ß $Ý Ý ß HÕ HÝ ã ß OÝ Ý !Ý 6Ý 9ß Ý +Ý -å $Ý 1Ý <÷ó õ +Ý 4ß 7ß vÕ v÷÷ ñ õ FÝ CÝ CÝ AÝ =Ý ?ß FÝ 3Ý 2Ý ?Ý A÷÷ ñ õ *÷ó õ 5ß BÝ ,Ý #Ý -Ý =Ý 3÷÷ ÷ ÷ ÷ õ ÷. jÑ iß dÓ d÷õ ÷ Kß HÝ 9ñ ×ÒÝ3Ý<çÝ'ð  %×0Ñ0×=Ò=Ó?Ð á×Òß3Ð3ÝFà '§£¨kÓ :¸g¿m»mÈFÓ>SÑ SÑà %Ðð 
×	Ó	˜HÓ	%€à�zŠz�~Š~˜n¨cÓ2×8Ò8Ó:€Ø—J’J—N’NÐ#6¸Ó<×BÒBÓDÐ Ù%Ð&CÐK\Ô]Ð Ø€ØÐ ñ �$Ô÷-ð -ó ð-õ6õ.ò`ð ñó ðð ñ#ó ð#ð ö0ó ð0ò.ò1ð  ˆE�JŠJØˆ%�+Š+Øˆ%�*Š*Ø	ˆ5�;Š;Ø	ˆ5�<Š<Ø	ˆ5�=Š=Ø
ˆE�NŠNØ	ˆ5�;Š;Ø	ˆ5�<Š<Ø	ˆ5�=Š=Ø	ˆ5�=Š=Ø	ˆ5�;Š;Ø	ˆ5�<Š<Øˆu×"Ò"Øˆu× Ò ðÐ õ&÷2/kõdõõ,õ>%õ*Mõ`(÷÷I.÷XP÷fjñ j÷Zj*ñ j*ôZl:+�b—i’iÐ!5Ð7GÈÐYiô l:+ñ^u (¨×(CÒ(CÓD€Ô Ø×Ò×&Ò&Ò2Ø*9×*EÒ*E×*MÒ*M×*TÒ*TØ [¸|ð +Uó +€O×ÒÔ'ð
 
Þ Yó 
Ø Yð 
Þ Mó 
Ø M÷ö2@÷ð öDFgôR"9Ð)õ "9òL /AÓ.BÑ Ó BôE ?÷ Er•   