+
    QV-j^  ã                   óÆ   € ^ RI Ht ^RIHt ]'       d   ^RIHt ^RIHt ^RIH	t	H
t
HtHt ^RIHt ]! 4       '       d   ^ RIt]P                   ! ]4      t ! R R	]4      tR# )
é    )ÚTYPE_CHECKING)ÚHfQuantizer)ÚPreTrainedModel)Ú
EetqConfig)Úis_accelerate_availableÚis_kernels_availableÚis_torch_availableÚlogging)Úget_module_from_nameNc                   óª   a a€ ] tR t^"t oRtRtV 3R ltR tV3R lR ltV3R lR lt	V3R	 lR
 lt
R t]V3R lR l4       tR tV3R ltRtVtV ;t# )ÚEetqHfQuantizerz2
8-bit quantization from EETQ quantization method
Fc                ó*   <€ \         SV `  ! V3/ VB  R # )N)ÚsuperÚ__init__)ÚselfÚquantization_configÚkwargsÚ	__class__s   &&,€Úw/Volumes/fast/ai/experiments/ui-tars-smoke/.venv/lib/python3.14/site-packages/transformers/quantizers/quantizer_eetq.pyr   ÚEetqHfQuantizer.__init__*   s   ø€ Ü‰ÒÐ,Ñ7°Ô7ó    c                óâ  € \        4       '       g   \        R 4      h\        4       '       g   \        R4      h\        P                  P                  4       '       g   \        R4      hVP                  R4      pVf   \        P                  R4       R# \        V\        4      '       dH   \        V4      ^8”  d   RVP                  4       9   g   RVP                  4       9   d   \        R4      hR# R# )	zHLoading an EETQ quantized model requires kernels (`pip install kernels`)zNLoading an EETQ quantized model requires accelerate (`pip install accelerate`)z/No GPU found. A GPU is needed for quantization.Ú
device_mapNzŽYou have loaded an EETQ model on CPU and have a CUDA device available, make sure to set your model on a GPU device in order to run your model.ÚcpuÚdiskz¯You are attempting to load an EETQ model with a device_map that contains a CPU or disk device. This is not supported. Please remove the CPU or disk device from the device_map.)r   ÚImportErrorr   ÚtorchÚcudaÚis_availableÚRuntimeErrorÚgetÚloggerÚwarning_onceÚ
isinstanceÚdictÚlenÚvaluesÚ
ValueError)r   Úargsr   r   s   &*, r   Úvalidate_environmentÚ$EetqHfQuantizer.validate_environment-   sÎ   € Ü#×%Ò%ÜÐhÓiÐiä&×(Ò(ÜÐnÓoÐoä�z‰z×&Ñ&×(Ò(ÜÐPÓQÐQà—Z‘Z Ó-ˆ
ØÒÜ×ÑðIöô ˜
¤D×)Ò)Ü�:‹ Ô" u°
×0AÑ0AÓ0CÔ'CÀvÐQ[×QbÑQbÓQdÔGdÜ ðhóð ñ Heñ *r   c                ó"   <€ V ^8„  d   QhRRRR/# )é   Údtypeztorch.dtypeÚreturn© )ÚformatÚ__classdict__s   "€r   Ú__annotate__ÚEetqHfQuantizer.__annotate__D   s   ø€ ÷ ñ  -ð °Mñ r   c                óZ   € V\         P                  8w  d   \        P                  R 4       V# )zLWe suggest you to set `dtype=torch.float16` for better efficiency with EETQ.)r   Úfloat16r"   Úinfo)r   r.   s   &&r   Úupdate_dtypeÚEetqHfQuantizer.update_dtypeD   s    € Ø”E—M‘MÔ!Ü�K‰KÐfÔgØˆr   c                ó*   <€ V ^8„  d   QhRRRS[ RS[/# )r-   Úmodelr   Ú
param_namer/   )ÚstrÚbool)r1   r2   s   "€r   r3   r4   I   s$   ø€ ÷ 
ñ 
Ð.?ð 
ÉSð 
Ñ_cñ 
r   c                óˆ   € ^RI Hp \        W4      w  rV\        WT4      '       d   V P                  '       g   VR8X  d   R# R# R# )r-   )Ú
EetqLinearÚbiasFT)Úintegrations.eetqr@   r   r$   Úpre_quantized)r   r;   r<   r   r@   ÚmoduleÚtensor_names   &&&,   r   Úparam_needs_quantizationÚ(EetqHfQuantizer.param_needs_quantizationI   s9   € Ý2ä2°5ÓEÑˆä�f×)Ò)Ø×!×!Ð! [°FÔ%:ÙáÙr   c                ó   <€ V ^8„  d   QhRR/# )r-   r;   r   r0   )r1   r2   s   "€r   r3   r4   U   s   ø€ ÷ 
ñ 
à ñ
r   c                ó¸   € ^RI Hp V P                  WP                  P                  VP
                  4      V n        V! WP                  V P                  R7      pR# )r-   )Úreplace_with_eetq_linear)Úmodules_to_not_convertrC   N)ÚintegrationsrJ   Úget_modules_to_not_convertr   rK   Ú_keep_in_fp32_modulesrC   )r   r;   r   rJ   s   &&, r   Ú$_process_model_before_weight_loadingÚ4EetqHfQuantizer._process_model_before_weight_loadingU   sO   € õ
 	<à&*×&EÑ&EØ×+Ñ+×BÑBÀE×D_ÑD_ó'
ˆÔ#ñ )Ø×*EÑ*EÐUY×UgÑUgô
Šr   c                ó   € R # ©Tr0   ©r   s   &r   Úis_serializableÚEetqHfQuantizer.is_serializabled   s   € Ùr   c                ó    <€ V ^8„  d   QhRS[ /# )r-   r/   )r>   )r1   r2   s   "€r   r3   r4   h   s   ø€ ÷ ñ ™dñ r   c                ó   € R # rR   r0   rS   s   &r   Úis_trainableÚEetqHfQuantizer.is_trainableg   s   € ár   c                ó   € ^RI Hp V! V 4      # )r-   )ÚEetqQuantize)rB   r[   )r   r[   s   & r   Úget_quantize_opsÚ EetqHfQuantizer.get_quantize_opsk   s   € Ý4á˜DÓ!Ð!r   c                ó$   <€ V ^8„  d   Qh/ R;R&   # )r-   r   r   r0   )r1   r2   s   "€r   r3   r4   "   s   ø‡ ‚ ð &Ñ%ò r   )rK   )Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__Úrequires_calibrationr   r*   r8   rF   rO   rT   ÚpropertyrX   r\   Ú__annotate_func__Ú__static_attributes__Ú__classdictcell__Ú__classcell__)r   r2   s   @@r   r   r   "   s`   ù‡ € ñð !Ðõ8ò÷.ð ÷

ð 
÷
ð 
òð ÷ó ðò"÷S … r   r   )Útypingr   Úbaser   Úmodeling_utilsr   Úutils.quantization_configr   Úutilsr   r   r	   r
   Úquantizers_utilsr   r   Ú
get_loggerr_   r"   r   r0   r   r   Ú<module>rq      sO   ðõ !å ÷ Ý0Ý6ç ^Ó ^Ý 2ñ ×ÒÛð 
×	Ò	˜HÓ	%€ôL"�kö L"r   