+
    QV-jÁ  ã                   óÆ   € ^ RI Ht ^RIHt ^RIHt ]'       d   ^RIHt ^RIH	t	H
t
HtHt ^RIHt ]! 4       '       d   ^ RIt]P                   ! ]4      t ! R R	]4      tR# )
é    )ÚTYPE_CHECKING)ÚHfQuantizer)Úget_module_from_name)ÚPreTrainedModel)Úis_accelerate_availableÚis_optimum_quanto_availableÚis_torch_availableÚlogging©ÚQuantoConfigNc                   óÎ   a a€ ] tR t^&t oRtRtV3R lV 3R lltR tV3R lR ltV3R lR	 lt	V3R
 lV 3R llt
V3R lR lt]V3R lR l4       tR tR tV3R ltRtVtV ;t# )ÚQuantoHfQuantizerz"
Quantizer for the quanto library
Fc                ó    <€ V ^8„  d   QhRS[ /# )é   Úquantization_configr   )ÚformatÚ__classdict__s   "€Úy/Volumes/fast/ai/experiments/ui-tars-smoke/.venv/lib/python3.14/site-packages/transformers/quantizers/quantizer_quanto.pyÚ__annotate__ÚQuantoHfQuantizer.__annotate__.   s   ø€ ÷ bñ b©Lñ bó    c                ó”   <€ \         SV `  ! V3/ VB  R ^R^RRRR/pVP                  V P                  P                  R4      V n        R# )Úint8Úfloat8Úint4g      à?Úint2g      Ð?N)ÚsuperÚ__init__Úgetr   ÚweightsÚquantized_param_size)Úselfr   ÚkwargsÚmap_to_param_sizeÚ	__class__s   &&, €r   r   ÚQuantoHfQuantizer.__init__.   sV   ø€ Ü‰ÒÐ,Ñ7°Ò7à�AØ�aØ�CØ�Dð	
Ðð %6×$9Ñ$9¸$×:RÑ:R×:ZÑ:ZÐ\`Ó$aˆÖ!r   c                ó�  € \        4       '       g   \        R 4      h\        4       '       g   \        R4      hVP                  R4      p\	        V\
        4      '       dF   \        V4      ^8”  d   RVP                  4       9   g   RVP                  4       9   d   \        R4      hV P                  P                  e   \        R4      hR# )zhLoading an optimum-quanto quantized model requires optimum-quanto library (`pip install optimum-quanto`)z`Loading an optimum-quanto quantized model requires accelerate library (`pip install accelerate`)Ú
device_mapÚcpuÚdiskzÜYou are attempting to load an model with a device_map that contains a CPU or disk device.This is not supported with quanto when the model is quantized on the fly. Please remove the CPU or disk device from the device_map.NzÂWe don't support quantizing the activations with transformers library.Use quanto library for more complex use cases such as activations quantization, calibration and quantization aware training.)r   ÚImportErrorr   r   Ú
isinstanceÚdictÚlenÚvaluesÚ
ValueErrorr   Úactivations)r"   Úargsr#   r(   s   &*, r   Úvalidate_environmentÚ&QuantoHfQuantizer.validate_environment8   s¿   € Ü*×,Ò,ÜØzóð ô '×(Ò(ÜØróð ð —Z‘Z Ó-ˆ
Ü�j¤$×'Ò'Ü�:‹ Ô" u°
×0AÑ0AÓ0CÔ'CÀvÐQ[×QbÑQbÓQdÔGdÜ ðPóð ð
 ×#Ñ#×/Ñ/Ò;ÜðOóð ñ <r   c                ó*   <€ V ^8„  d   QhRRRS[ RS[/# )r   Úmodelr   Ú
param_nameÚreturn)ÚstrÚbool)r   r   s   "€r   r   r   O   s$   ø€ ÷ 	ñ 	Ð.?ð 	ÉSð 	Ñ_cñ 	r   c                ó~   € ^ RI Hp \        W4      w  rV\        WT4      '       d   RV9   d   VP                  '       * # R# )r   )ÚQModuleMixinÚweightF)Úoptimum.quantor<   r   r,   Úfrozen)r"   r6   r7   r#   r<   ÚmoduleÚtensor_names   &&&,   r   Úparam_needs_quantizationÚ*QuantoHfQuantizer.param_needs_quantizationO   s4   € Ý/ä2°5ÓEÑˆä�f×+Ò+°¸KÔ0Gà—}‘}Ô$Ð$ár   c                ór   <€ V ^8„  d   QhRS[ S[S[S[,          3,          RS[ S[S[S[,          3,          /# )r   Ú
max_memoryr8   )r-   r9   Úint)r   r   s   "€r   r   r   Z   s6   ø€ ÷ ñ ©D±±c¹Cµi°Õ,@ð ÁTÉ#ÉsÑUXÍyÈ.ÕEYñ r   c                óf   € VP                  4        UUu/ uF  w  r#W#R ,          bK  	  pppV# u uppi )gÍÌÌÌÌÌì?)Úitems)r"   rE   ÚkeyÚvals   &&  r   Úadjust_max_memoryÚ#QuantoHfQuantizer.adjust_max_memoryZ   s5   € Ø6@×6FÑ6FÔ6HÔIÑ6H©(¨#�c �:’oÑ6Hˆ
ÑIØÐùó Js   ”-c                ó.   <€ V ^8„  d   QhRRRS[ RRRS[/# )r   r6   r   r7   Úparamztorch.Tensorr8   )r9   Úfloat)r   r   s   "€r   r   r   ^   s2   ø€ ÷ Dñ DÐ(9ð DÁsð DÐSað DÑfkñ Dr   c                ó†   <€ V P                  W4      '       d   V P                  e   V P                  # \        SV `  WV4      # )z4Return the element size (in bytes) for `param_name`.)rB   r!   r   Úparam_element_size)r"   r6   r7   rN   r%   s   &&&&€r   rQ   Ú$QuantoHfQuantizer.param_element_size^   s=   ø€ à×(Ñ(¨×;Ò;À×@YÑ@YÒ@eØ×,Ñ,Ð,ä‰wÑ)¨%¸UÓCÐCr   c                ó   <€ V ^8„  d   QhRR/# )r   r6   r   © )r   r   s   "€r   r   r   e   s   ø€ ÷ 	
ñ 	
Ð:Kñ 	
r   c                ó¸   € ^RI Hp V P                  WP                  P                  VP
                  4      V n        V! WP                  V P                  R7      pR# )r   )Úreplace_with_quanto_layers)Úmodules_to_not_convertr   N)ÚintegrationsrV   Úget_modules_to_not_convertr   rW   Ú_keep_in_fp32_modules)r"   r6   r#   rV   s   &&, r   Ú$_process_model_before_weight_loadingÚ6QuantoHfQuantizer._process_model_before_weight_loadinge   sM   € Ý=à&*×&EÑ&EØ×+Ñ+×BÑBÀE×D_ÑD_ó'
ˆÔ#ñ +Ø×*EÑ*EÐ[_×[sÑ[sô
Šr   c                ó    <€ V ^8„  d   QhRS[ /# )r   r8   )r:   )r   r   s   "€r   r   r   q   s   ø€ ÷ ñ ™dñ r   c                ó   € R # )TrT   ©r"   s   &r   Úis_trainableÚQuantoHfQuantizer.is_trainablep   s   € ár   c                ó   € R # )FrT   r_   s   &r   Úis_serializableÚ!QuantoHfQuantizer.is_serializablet   s   € Ùr   c                ó   € ^RI Hp V! V 4      # )r   )ÚQuantoQuantize)Úintegrations.quantorf   )r"   rf   s   & r   Úget_quantize_opsÚ"QuantoHfQuantizer.get_quantize_opsw   s   € Ý8á˜dÓ#Ð#r   c                ó$   <€ V ^8„  d   Qh/ R;R&   # )r   r   r   rT   )r   r   s   "€r   r   r   &   s   ø‡ ‚ ð (Ñ'ò r   )rW   r!   )Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__Úrequires_calibrationr   r3   rB   rK   rQ   r[   Úpropertyr`   rc   rh   Ú__annotate_func__Ú__static_attributes__Ú__classdictcell__Ú__classcell__)r%   r   s   @@r   r   r   &   ss   ù‡ € ñð !Ð÷bó bò÷.	ð 	÷ð ÷Dó D÷	
ð 	
ð ÷ó ðòò$÷c … r   r   )Útypingr   Úbaser   Úquantizers_utilsr   Úmodeling_utilsr   Úutilsr   r   r	   r
   Úutils.quantization_configr   ÚtorchÚ
get_loggerrk   Úloggerr   rT   r   r   Ú<module>r      sS   ðõ !å Ý 2÷ Ý0÷ó õ 5ñ ×ÒÛà	×	Ò	˜HÓ	%€ôT$˜ö T$r   