+
    QV-j(5  ã                   ó\  € ^ RI HtHt ^ RIHtHt ^RIHtHt ^RI	H
t
Ht ^RIHt ]'       d   ^ RIHt ^RIHt ]! 4       '       d   ^ RIt]'       g   ^ RIHt M]t]P(                  ! ]4      tR	 R
 ltR t ! R R]4      t ! R R]4      tRR]R]P6                  ]P8                  .//tR# )é    )ÚABCÚabstractmethod)ÚTYPE_CHECKINGÚAny)Úis_torch_availableÚlogging)ÚQuantizationConfigMixinÚQuantizationMethod)Úget_module_from_name)Ú
ModuleList©ÚPreTrainedModelNc                ó$   € V ^8„  d   QhR\         /# ©é   Úreturn)Úlist)Úformats   "Úm/Volumes/fast/ai/experiments/ui-tars-smoke/.venv/lib/python3.14/site-packages/transformers/quantizers/base.pyÚ__annotate__r   &   s   € ÷ (ñ (¤dñ (ó    c                ót  € \        4       p\        V P                  4      ^ 8”  dL   \        V P                  P                  4       4      \        V P                  P	                  4       4      ,          p\        V P                  4       4      R,          ^ ,          0pV P                  4       pV P                  4        UUu0 uF(  w  rEVf   K  \        V4      \        V4      8X  g   K&  VkK*  	  pppW,          V,          p\        V Uu0 uF  qˆP                  R4      kK  	  up4      p\        V4      # u uppi u upi )zÇ
Function to automatically detect keys to not convert for usage like quantization. For example for CausalLM modules
we may want to keep the lm_head in full precision for numerical stability reasons.
z.weightéÿÿÿÿ)ÚsetÚlenÚall_tied_weights_keysÚvaluesÚkeysr   Únamed_parametersÚget_output_embeddingsÚnamed_modulesÚidÚremovesuffix)	ÚmodelÚ	tied_keysÚlast_module_keyÚoutput_emb_moduleÚnameÚmoduleÚoutput_emb_keysÚmodules_to_not_convertÚks	   &        r   Úget_keys_to_not_convertr-   &   s  € ô “€IÜ
ˆ5×&Ñ&Ó'¨!Ô+Ü˜×3Ñ3×:Ñ:Ó<Ó=ÄÀE×D_ÑD_×DdÑDdÓDfÓ@gÕgˆ	ô ˜E×2Ñ2Ó4Ó5°bÕ9¸!Õ<Ð=€Oð ×3Ñ3Ó5Ðð "×/Ñ/Ô1ôá1‰LˆDØô 	ä-/°«Z¼2Ð>OÓ;PÑ-P÷ 	ˆÙ1ð ñ ð
 'Õ8¸?ÕJÐä!ÑF\Ó"]ÑF\À§>¡>°)Ö#<ÑF\Ñ"]Ó^ÐäÐ&Ó'Ð'ùóùò #^s   Â;D/ÃD/Ã#D/ÄD5c                 óˆ   € ^RI Hp V P                  4        F'  p\        W!4      '       g   K  RVP                  n        K)  	  R# )r   r   TN)Úmodeling_utilsr   ÚmodulesÚ
isinstanceÚconfigÚ_is_quantized)r$   r   r)   s   &  r   Ú_assign_is_quantizedr4   A   s,   € Ý0à—-‘-–/ˆÜ�f×.Ô.Ø*.ˆF�M‰MÖ'ó "r   c                   ó²  a € ] tR t^It o RtRtV 3R lR ltV 3R lR ltV 3R lR ltV 3R	 lR
 lt	V 3R lR lt
V 3R lR ltR tR tR tR tR-V 3R lR lltV 3R lR ltV 3R lR ltR tR-R ltR-R ltV 3R lR lt]R.V 3R lR  ll4       t]V 3R! lR" l4       t]V 3R# lR$ l4       tR% t]R& 4       t]]R' 4       4       tR( t R) t!R* t"R+ t#R,t$V t%R# )/ÚHfQuantizerac  
Abstract class of the HuggingFace quantizer. Supports for now quantizing HF transformers models for inference and/or quantization.
This class is used only for transformers.PreTrainedModel.from_pretrained and cannot be easily used outside the scope of that method
yet.

Attributes
    quantization_config (`transformers.utils.quantization_config.QuantizationConfigMixin`):
        The quantization config that defines the quantization parameters of your model that you want to quantize.
    requires_calibration (`bool`):
        Whether the quantization method requires to calibrate the model before using it.
Fc                ó    <€ V ^8„  d   QhRS[ /# )r   Úquantization_config)r	   )r   Ú__classdict__s   "€r   r   ÚHfQuantizer.__annotate__X   s   ø€ ÷ 	ñ 	Ñ,Cñ 	r   c                ó¾   € Wn         VP                  R R4      V n        V P                  '       g.   V P                  '       d   \	        RVP
                   R24      hR# R# )Úpre_quantizedTzThe quantization method zÏ does require the model to be pre-quantized. You explicitly passed `pre_quantized=False` meaning your model weights are not quantized. Make sure to pass `pre_quantized=True` while knowing what you are doing.N)r8   Úpopr<   Úrequires_calibrationÚ
ValueErrorÚquant_method)Úselfr8   Úkwargss   &&,r   Ú__init__ÚHfQuantizer.__init__X   sd   € Ø#6Ô Ø#ŸZ™Z¨¸Ó>ˆÔà×!×!Ð! d×&?×&?Ð&?ÜØ*Ð+>×+KÑ+KÐ*Lð MNð Oóð ñ '@Ñ!r   c                ó"   <€ V ^8„  d   QhRRRR/# )r   Údtypeztorch.dtyper   © )r   r9   s   "€r   r   r:   c   s   ø€ ÷ 
ñ 
 -ð 
°Mñ 
r   c                ó   € V# )a  
Some quantization methods require to explicitly set the dtype of the model to a
target dtype. You need to override this method in case you want to make sure that behavior is
preserved

Args:
    dtype (`torch.dtype`):
        The input dtype that is passed in `from_pretrained`
rG   )rA   rF   s   &&r   Úupdate_dtypeÚHfQuantizer.update_dtypec   s	   € ð ˆr   c                ón   <€ V ^8„  d   QhRS[ S[S[3,          R,          RS[ S[S[3,          R,          /# )r   Ú
device_mapNr   )ÚdictÚstrr   )r   r9   s   "€r   r   r:   o   s7   ø€ ÷ 
ñ 
©D±±c°­N¸TÕ,Að 
ÁdÉ3ÑPSÈ8ÅnÐW[ÕF[ñ 
r   c                ó   € V# )aa  
Override this method if you want to pass a override the existing device map with a new
one. E.g. for bitsandbytes, since `accelerate` is a hard requirement, if no device_map is
passed, the device_map is set to `"auto"``

Args:
    device_map (`Union[dict, str]`, *optional*):
        The device_map that is passed through the `from_pretrained` method.
rG   )rA   rL   s   &&r   Úupdate_device_mapÚHfQuantizer.update_device_mapo   s
   € ð Ðr   c                ó.   <€ V ^8„  d   QhRRRS[ RRRS[/# )r   r$   r   Ú
param_nameÚparamútorch.Tensorr   )rN   Úfloat)r   r9   s   "€r   r   r:   {   s,   ø€ ÷ $ñ $Ð(9ð $Ásð $ÐSað $Ñfkñ $r   c                ó"   € VP                  4       # ©N)Úelement_size)rA   r$   rS   rT   s   &&&&r   Úparam_element_sizeÚHfQuantizer.param_element_size{   s   € Ø×!Ñ!Ó#Ð#r   c                ór   <€ V ^8„  d   QhRS[ S[S[S[,          3,          RS[ S[S[S[,          3,          /# )r   Ú
max_memoryr   )rM   rN   Úint)r   r9   s   "€r   r   r:   ~   s6   ø€ ÷ ñ ©D±±c¹Cµi°Õ,@ð ÁTÉ#ÉsÑUXÍyÈ.ÕEYñ r   c                ó   € V# )zaadjust max_memory argument for infer_auto_device_map() if extra memory is needed for quantizationrG   )rA   r]   s   &&r   Úadjust_max_memoryÚHfQuantizer.adjust_max_memory~   s   € àÐr   c                ó*   <€ V ^8„  d   QhRRRS[ RS[/# )r   r$   r   rS   r   )rN   Úbool)r   r9   s   "€r   r   r:   ‚   s$   ø€ ÷ ñ Ð.?ð ÉSð Ñ_cñ r   c                ó   € R# )z4
Check whether a given param needs to be quantized.
FrG   )rA   r$   rS   rB   s   &&&,r   Úparam_needs_quantizationÚ$HfQuantizer.param_needs_quantization‚   s   € ñ r   c                ó   € R# )a  
This method is used to potentially check for potential conflicts with arguments that are
passed in `from_pretrained`. You need to define it for all future quantizers that are integrated with transformers.
If no explicit check are needed, simply return nothing.
NrG   )rA   ÚargsrB   s   &*,r   Úvalidate_environmentÚ HfQuantizer.validate_environmentˆ   s   € ñ 	r   c                ó   € V# ©z"updates the tp plan for the scalesrG   ©rA   r2   s   &&r   Úupdate_tp_planÚHfQuantizer.update_tp_plan�   ó   € àˆr   c                ó   € V# rl   rG   rm   s   &&r   Úupdate_ep_planÚHfQuantizer.update_ep_plan”   rp   r   c                ó   € V# rX   rG   ©rA   r$   rB   s   &&,r   Ú$_process_model_before_weight_loadingÚ0HfQuantizer._process_model_before_weight_loading˜   ó   € Øˆr   Nc                ó   <€ V ^8„  d   QhRR/# ©r   r$   r   rG   )r   r9   s   "€r   r   r:   ›   s   ø€ ÷ Cñ CÐ&7ñ Cr   c                óÎ   € \        VRR4       \        VRV P                  P                  4       V P                  '       d   V P	                  V4       V P
                  ! V3/ VB  R# )a	  
Setting model attributes and/or converting model before weights loading. At this point
the model should be initialized on the meta device so you can freely manipulate the skeleton
of the model in order to replace modules in-place. Make sure to override the abstract method `_process_model_before_weight_loading`.

Args:
    model (`~transformers.PreTrainedModel`):
        The model to quantize
    kwargs (`dict`, *optional*):
        The keyword arguments that are passed along `_process_model_before_weight_loading`.
Úis_quantizedTÚquantization_methodN)Úsetattrr8   r@   r<   Ú_convert_model_for_quantizationrv   )rA   r$   rF   rB   s   &&&,r   Úpreprocess_modelÚHfQuantizer.preprocess_model›   sV   € ô 	��~ tÔ,Ü�Ð,¨d×.FÑ.F×.SÑ.SÔTØ××ÐØ×0Ñ0°Ô7Ø×1Ò1°%ÑB¸6ÔBr   c                ó   <€ V ^8„  d   QhRR/# rz   rG   )r   r9   s   "€r   r   r:   ­   s   ø€ ÷ ñ Ð9Jñ r   c                ó   € V# rX   rG   ru   s   &&,r   Ú#_process_model_after_weight_loadingÚ/HfQuantizer._process_model_after_weight_loading­   rx   r   c                ó   <€ V ^8„  d   QhRR/# rz   rG   )r   r9   s   "€r   r   r:   °   s   ø€ ÷ Iñ IÐ'8ñ Ir   c                óö   € V P                   VP                  n         V P                  '       d0   \        V P                   RR4      '       d   V P	                  V4       M\        V4       V P                  ! V3/ VB # )aM  
Post-process the model post weights loading.
Make sure to override the abstract method `_process_model_after_weight_loading`.

Args:
    model (`~transformers.PreTrainedModel`):
        The model to quantize
    kwargs (`dict`, *optional*):
        The keyword arguments that are passed along `_process_model_after_weight_loading`.
Ú
dequantizeF)r8   r2   r<   ÚgetattrÚremove_quantization_configr4   r„   ru   s   &&,r   Úpostprocess_modelÚHfQuantizer.postprocess_model°   sc   € ð ,0×+CÑ+Cˆ�‰Ô(à××Ð¤'¨$×*BÑ*BÀLÐRW×"XÒ"XØ×+Ñ+¨EÕ2ä  Ô'à×7Ò7¸ÑHÀÑHÐHr   c                ó´   € \        VR4      '       d   V=\        VP                  R4      '       d   VP                  =\        VR4      '       d   V=RVn        R# )z0
Remove the quantization config from the model.
Úhf_quantizerr8   r}   FN)ÚhasattrrŽ   r2   r8   r}   r|   ©rA   r$   s   &&r   rŠ   Ú&HfQuantizer.remove_quantization_configÄ   sO   € ô �5˜.×)Ò)ØÐ"Ü�5—<‘<Ð!6×7Ò7Ø—‘Ð0Ü�5Ð/×0Ò0ØÐ)Ø"ˆÖr   c                ó€   € Vf   VP                   P                  pV P                  WR7      pV P                  V4       V# )zœ
Potentially dequantize the model to retrieve the original model, with some loss in accuracy / performance.
Note not all quantization schemes support this.
)rF   )r2   rF   Ú_dequantizerŠ   ©rA   r$   rF   s   &&&r   rˆ   ÚHfQuantizer.dequantizeÐ   s@   € ð
 Š=ð —L‘L×&Ñ&ˆEØ× Ñ  Ð Ó4ˆØ×'Ñ'¨Ô.àˆr   c                óF   € \        V P                  P                   R 24      h)zH has no implementation of `dequantize`, please raise an issue on GitHub.©ÚNotImplementedErrorr8   r@   r”   s   &&&r   r“   ÚHfQuantizer._dequantizeÞ   s'   € Ü!Ø×'Ñ'×4Ñ4Ð5Ð5}Ð~ó
ð 	
r   c                ó&   <€ V ^8„  d   QhRS[ RS[ /# )r   rS   r   )rN   )r   r9   s   "€r   r   r:   ã   s   ø€ ÷ ñ ©ð ±ñ r   c                ó   € V# )z>
Override this method if you want to adjust the `param_name`.
rG   )rA   rS   s   &&r   Úget_param_nameÚHfQuantizer.get_param_nameã   s
   € ð Ðr   c                ól   <€ V ^8„  d   QhRRRS[ S[,          R,          RS[ S[,          R,          RS[/# )r   r$   r   Úskip_modulesNÚkeep_in_fp32_modulesÚadd_default_skips)r   rN   rc   )r   r9   s   "€r   r   r:   ê   sC   ø€ ÷ &ñ &Ø ð&á™3•i $Õ&ð&ñ #¡3�i¨$Õ.ð&ñ  ñ	&r   c                ó¶   € Ve	   V'       d   \        V 4      pM. pVe   VP                  V4       Ve   VP                  V4       \        \        V4      4      pV# rX   )r-   Úextendr   r   )r$   rŸ   r    r¡   r+   s   &&&& r   Úget_modules_to_not_convertÚ&HfQuantizer.get_modules_to_not_converté   s^   € ð Ò×#4Ü%<¸UÓ%CÑ"à%'Ð"àÒ#Ø"×)Ñ)¨,Ô7àÒ+Ø"×)Ñ)Ð*>Ô?ä!%¤cÐ*@Ó&AÓ!BÐà%Ð%r   c                ó    <€ V ^8„  d   QhRS[ /# r   ©rc   )r   r9   s   "€r   r   r:      s   ø€ ÷ ñ ¡$ñ r   c                ó   € R# )zUFlag indicating whether the quantized model can carry out quantization aware trainingFrG   ©rA   s   &r   Úis_qat_trainableÚHfQuantizer.is_qat_trainableÿ   ó   € ñ r   c                ó    <€ V ^8„  d   QhRS[ /# r   r§   )r   r9   s   "€r   r   r:     s   ø€ ÷ ñ ¡ñ r   c                ó   € R# )z;Flag indicating whether the quantized model can be compiledFrG   r©   s   &r   Úis_compileableÚHfQuantizer.is_compileable  r¬   r   c                ó
   € R/ 3# )zcGet state dict and metadata. Useful when we need to modify a bit the state dict due to quantizationNrG   r�   s   &&r   Úget_state_dict_and_metadataÚ'HfQuantizer.get_state_dict_and_metadata	  s   € à�Rˆxˆr   c                ó   € R # rX   rG   r©   s   &r   Úis_serializableÚHfQuantizer.is_serializable  s   € Ù"r   c                ó   € R # rX   rG   r©   s   &r   Úis_trainableÚHfQuantizer.is_trainable  s   € ár   c                óê  € VP                  4        FÊ  w  r#VP                  P                  pV\        9   g   K(  V P                  P
                  \        V,          R ,          9   g   KW  \        P                  ! R4      ;_uu_ 4        \        W4      w  rR\        V,          R,          ! VP                  P                  4       4      VP                  V&   RRR4       KÌ  	  R#   + '       g   i     Ká  ; i)Úquantization_methodsÚmetaÚmodule_nameN)r!   Ú	__class__Ú__name__Ú!MODULES_TO_PATCH_FOR_QUANTIZATIONr8   r@   ÚtorchÚdevicer   r2   Úget_text_configÚ_modules)rA   r$   r(   r)   Úmodule_class_nameÚparent_modules   &&    r   r   Ú+HfQuantizer._convert_model_for_quantization  s¹   € Ø!×/Ñ/Ö1‰LˆDØ &× 0Ñ 0× 9Ñ 9ÐØ Ô$EÖEØ×(Ñ(×5Ñ5Ü4Ð5FÕGÐH^Õ_ö`ô —\’\ &×)Õ)Ü*>¸uÓ*KÑ'�MÜ3TÐUfÕ3gÐhuÖ3vØŸ™×4Ñ4Ó6ó4�M×*Ñ*¨4Ñ0÷ *Ñ)ó 2÷ *×)Ð)ús   ÂAC!Ã!C2c                óF   € \        V P                  P                   R 24      h)z1 is not available yet and will be supported soon.r—   r©   s   &r   Úget_quantize_opsÚHfQuantizer.get_quantize_ops!  s'   € Ü!Ø×'Ñ'×4Ñ4Ð5Ð5fÐgó
ð 	
r   c                ó   € . # rX   rG   r©   s   &r   Úget_weight_conversionsÚ"HfQuantizer.get_weight_conversions&  s   € Øˆ	r   c                ó.   € WP                  4       ,           # )uH  Give the quantizer a chance to rewrite the weight conversion pipeline.

Loading runs ``renamings â†’ converters â†’ (dequant â†’ merge â†’ concat)``. Dequant
has to happen *before* any merge/concat op because those operations aren't
aware of per-block scales, so the per-expert (weight, scale) pairs need to be
collapsed into full-precision tensors first. Subclasses (e.g. the FP8
quantizer in ``dequantize=True`` mode) override this to inject a dequantize
op at the start of each model-provided :class:`WeightConverter` and attach the
matching scale source patterns. Default: no-op.
)rÌ   )rA   Úweight_conversionss   &&r   Úupdate_weight_conversionsÚ%HfQuantizer.update_weight_conversions)  s   € ð "×$?Ñ$?Ó$AÕAÐAr   )r<   r8   rX   )NNF)&r¿   Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__r>   rC   rI   rP   rZ   r`   re   ri   rn   rr   rv   r€   r„   r‹   rŠ   rˆ   r“   rœ   Ústaticmethodr¤   Úpropertyrª   r¯   r²   r   rµ   r¸   r   rÉ   rÌ   rÐ   Ú__static_attributes__Ú__classdictcell__)r9   s   @r   r6   r6   I   s$  ø‡ € ñ
ð !Ð÷	ð 	÷
ð 
÷
ð 
÷$ð $÷ð ÷ð òòòò÷Cò C÷$ð ÷Ið Iò(
#ôô
÷
ð ð ÷&ñ &ó ð&ð* ÷ó ðð ÷ó ðòð Ù"ó Ø"àØÙó ó àòò
ò
÷Bð Br   r6   c                   óH   a a€ ] tR tRt oRtV 3R ltV3R lR ltRtVtV ;t	# )ÚSequentialLlama4TextExpertsi7  z˜
A module that implements a compressed version of a list of expert modules.
This is specifically designed to work with Llama4TextExperts in MoE layers.
c                ó®   <€ ^ RI Hp \        ST `  \	        VP
                  4       Uu. uF
  q2! V4      NK  	  up4       VP
                  V n        R# u upi )r   )ÚLlama4TextMLPN)Ú*transformers.models.llama4.modeling_llama4rÝ   ÚsuperrC   ÚrangeÚnum_local_expertsÚnum_experts)rA   r2   rÝ   Ú_r¾   s   &&  €r   rC   Ú$SequentialLlama4TextExperts.__init__=  sH   ø€ ÝLä‰Ñ¼¸v×?WÑ?WÔ9XÓYÑ9X°A˜-¨Ö/Ñ9XÑYÔZØ!×3Ñ3ˆÖùò Zs   ¨Ac                ó"   <€ V ^8„  d   QhRRRR/# )r   Úhidden_statesrU   r   rG   )r   r9   s   "€r   r   Ú(SequentialLlama4TextExperts.__annotate__C  s   ø€ ÷ ñ à%ðð 
ñr   c                óò   € VP                  V P                  RVP                  R,          4      p\        P                  ! V4      p\        V P                  4       F  pW,          ! W,          4      W#&   K  	  V# )é   r   )Úreshaperâ   ÚshaperÁ   Ú
zeros_likerà   )rA   ræ   Ú
routed_outÚ
expert_idxs   &&  r   ÚforwardÚ#SequentialLlama4TextExperts.forwardC  sh   € ð &×-Ñ-¨d×.>Ñ.>ÀÀM×DWÑDWÐXZÕD[Ó\ˆÜ×%Ò% mÓ4ˆ
Ü × 0Ñ 0Ö1ˆJØ%)Ö%5°mÕ6OÓ%PˆJÓ"ñ 2àÐr   )râ   )
r¿   rÒ   rÓ   rÔ   rÕ   rC   rï   rØ   rÙ   Ú__classcell__)r¾   r9   s   @@r   rÛ   rÛ   7  s   ù‡ € ñõ
4÷÷ ð r   rÛ   ÚLlama4TextExpertsr½   r»   )Úabcr   r   Útypingr   r   Úutilsr   r   Úutils.quantization_configr	   r
   Úquantizers_utilsr   Útorch.nnr   r/   r   rÁ   rN   Ú
get_loggerÚ__file__Úloggerr-   r4   r6   rÛ   ÚCOMPRESSED_TENSORSÚBITS_AND_BYTESrÀ   rG   r   r   Ú<module>rþ      s¢   ð÷ $ß %ç /ß SÝ 2÷ Ý#å0á×ÒÛçÝ'øà€Jà	×	Ò	˜HÓ	%€õ(ò6/ôkB�#ô kBô\ *ô ð0 ØÐ2ØØ×1Ñ1Ø×-Ñ-ð!
ðð%Ò !r   