+
    QV-jÙ3  ã                   óÚ   € ^ RI Ht ^RIHt ]'       d   ^RIHt ^RIHt ^RIH	t	H
t
HtHtHt ^RIHt ]! 4       '       d   ^ RIt^RIHt ]P&                  ! ]4      tRt ! R	 R
]4      tR# )é    )ÚTYPE_CHECKING)ÚHfQuantizer)ÚPreTrainedModel)ÚMxfp4Config)Úis_accelerate_availableÚis_kernels_availableÚis_torch_availableÚis_triton_availableÚlogging)Úget_module_from_nameN)ÚWeightConverterc                   óÌ   a a€ ] tR t^*t oRtRtV 3R ltR tR tV3R lR lt	V3R lR	 lt
RV3R
 lR lltR tR tR tR t]V3R lR l4       tR tR tV3R ltRtVtV ;t# )ÚMxfp4HfQuantizerz'
FP4 quantization using fbgemm kernels
Fc                ó8   <€ \         SV `  ! V3/ VB  R V n        R # ©N)ÚsuperÚ__init__Útriton_kernels_hub)ÚselfÚquantization_configÚkwargsÚ	__class__s   &&,€Úx/Volumes/fast/ai/experiments/ui-tars-smoke/.venv/lib/python3.14/site-packages/transformers/quantizers/quantizer_mxfp4.pyr   ÚMxfp4HfQuantizer.__init__2   s   ø€ Ü‰ÒÐ,Ñ7°Ò7Ø"&ˆÖó    c                óª   € V P                   f!    ^RIHp V! R4      V n         V P                   # V P                   #   \         d    \        R4      hi ; i)z3Lazy import and initialize kernels only when needed)Ú
get_kernelz(kernels-community/gpt-oss-triton-kernelsz2kernels package is required for MXFP4 quantization)r   Úintegrations.hub_kernelsr   ÚImportError)r   r   s   & r   Ú_lazy_import_kernelsÚ%Mxfp4HfQuantizer._lazy_import_kernels6   s]   € à×"Ñ"Ò*ðXÝAá*4Ð5_Ó*`�Ô'ð ×&Ñ&Ð&ˆt×&Ñ&Ð&øô ô XÜ!Ð"VÓWÐWðXús	   �; »Ac                ó4  € \        4       '       g   \        R 4      hV P                  P                  '       d   R# \	        4       '       g   \        R4      h\
        P                  P                  4       ;'       g    \
        P                  ! R4      pVP                  R9  dN   V P                  '       d-   \        P                  RV R24       RV P                  n        R# \        RV R24      h\
        P                  P                  4       '       d   Rp\!        R	4      p\#        4       pMŒ\
        P$                  P                  4       '       d:   \
        P$                  P'                  4       pVR8¬  p\!        R
4      p\#        4       pM/VP                  R8X  d   Rp\!        R	4      p\#        4       pMRpRpRpV P                  '       d’   V'       g)   \        P                  R4       RV P                  n        R# V'       g)   \        P                  R4       RV P                  n        R# V'       g)   \        P                  R4       RV P                  n        R# M9V'       g   \)        R4      hV'       g   \)        R4      hV'       g   \)        R4      hV P                  '       g   V P+                  4        VP-                  R4      pVeO   \/        V\0        4      '       d7   V P                  '       g#   RVP3                  4       9   d   \)        R4      hR# R# R# R# )zqUsing mxfp4 quantization requires torchPlease install the latest version of torch ( pip install --upgrade torch )Nz9Using mxfp4 requires Accelerate: `pip install accelerate`ÚcpuzGUsing MXFP4 quantized models requires model on cuda/xpu/cpu, but found zj, we will default to dequantizing the model to bf16. To use mxfp4, please disable the current accelerator.TzIQuantizing a model using MXFP4 requires model on cuda/xpu/cpu, but found z7. To use mxfp4, please disable the current accelerator.z3.5.0z3.4.0FuÒ   MXFP4 quantization is only supported on GPUs with compute capability >= 7.5 (e.g T4, A100, L4, H100, or B200) or XPUs (e.g IntelÂ® Data Center GPU Max Series). We will default to dequantizing the model to bf16.zÄMXFP4 quantization requires Triton: CUDA requires Triton >= 3.4.0, XPU/CPU requires Triton >= 3.5.0. Please install triton: `pip install triton`. We will default to dequantizing the model to bf16.z„MXFP4 quantization requires the `kernels` package: `pip install kernels>=0.12.0`. We will default to dequantizing the model to bf16.u¥   MXFP4 quantization is only supported on GPUs with compute capability >= 7.5 (e.g T4, A100, L4, H100, or B200) or XPUs (e.g IntelÂ® Data Center GPU Max Series) or CPUz�MXFP4 quantization requires Triton: CUDA requires Triton >= 3.4.0, XPU/CPU requires Triton >= 3.5.0. Please install triton: `pip install triton`zPMXFP4 quantization requires the `kernels` package: `pip install kernels>=0.12.0`Ú
device_mapÚdiskzäYou are attempting to load an FP4 model with a device_map that contains a disk device.This is not supported when the model is quantized on the fly. Please use a quantized checkpoint or remove the disk device from the device_map.)ÚcudaÚxpur#   )é   é   )r	   r   r   Ú
dequantizer   ÚtorchÚacceleratorÚcurrent_acceleratorÚdeviceÚtypeÚpre_quantizedÚloggerÚwarning_onceÚRuntimeErrorr'   Úis_availabler
   r   r&   Úget_device_capabilityÚ
ValueErrorr    ÚgetÚ
isinstanceÚdictÚvalues)	r   Úargsr   r.   Úis_device_supported_mxfp4Útriton_availableÚkernels_installedÚcompute_capabilityr$   s	   &*,      r   Úvalidate_environmentÚ%Mxfp4HfQuantizer.validate_environmentA   s×  € Ü!×#Ò#Üð]óð ð
 ×#Ñ#×.×.Ð.Ùä&×(Ò(ÜÐYÓZÐZä×"Ñ"×6Ñ6Ó8×OÐO¼E¿LºLÈÓ<OˆØ�;‰;Ð4Ô4Ø×!×!Ð!Ü×#Ñ#Ø]Ð^dÐ]eð  fPð  Qôð 7;�×(Ñ(Ô3Ùä"Ø_Ð`fÐ_gð  h_ð  `óð ô �9‰9×!Ñ!×#Ò#Ø(,Ð%Ü2°7Ó;ÐÜ 4Ó 6ÑÜ�Z‰Z×$Ñ$×&Ò&Ü!&§¡×!AÑ!AÓ!CÐØ(:¸fÑ(DÐ%Ü2°7Ó;ÐÜ 4Ó 6ÑØ�[‰[˜EÔ!Ø(,Ð%Ü2°7Ó;ÐÜ 4Ó 6Ñà(-Ð%Ø$ÐØ %Ðà××Ðß,Ü×#Ñ#ðIôð
 7;�×(Ñ(Ô3Ùç#Ü×#Ñ#ðIôð
 7;�×(Ñ(Ô3Ùç$Ü×#Ñ#ðIôð
 7;�×(Ñ(Ô3Ùð %÷ +Üðlóð ÷ "Üð`óð ÷ #ÜÐoÓpÐpà×!×!Ð!Ø×%Ñ%Ô'à—Z‘Z Ó-ˆ
ØÒ!¤j°¼T×&BÒ&BØ×%×%Ð%¨&°J×4EÑ4EÓ4GÔ*GÜ ðgóð ñ +HÑ%ñ 'CÑ!r   c                ó*   <€ V ^8„  d   QhRRRS[ RS[/# )é   Úmodelr   Ú
param_nameÚreturn)ÚstrÚbool)ÚformatÚ__classdict__s   "€r   Ú__annotate__ÚMxfp4HfQuantizer.__annotate__¡   s$   ø€ ÷ ñ Ð.?ð ÉSð Ñ_cñ r   c                ód   € ^RI Hp \        W4      w  rV\        WT4      '       d   VR9   d   R# R# R# )rC   ©ÚMxfp4GptOssExpertsFT)Údown_proj_biasÚgate_up_proj_bias)ÚintegrationsrO   r   r8   )r   rD   rE   r   rO   ÚmoduleÚtensor_names   &&&,   r   Úparam_needs_quantizationÚ)Mxfp4HfQuantizer.param_needs_quantization¡   s/   € Ý5ä2°5ÓEÑˆÜ�f×1Ò1ØÐEÔEÙÙÙr   c                ó   <€ V ^8„  d   QhRR/# )rC   rD   r   © )rI   rJ   s   "€r   rK   rL   «   s   ø€ ÷ $ñ $Ð9Jñ $r   c                ó  € \         P                  P                  4       '       d!   \         P                  P                  4        R # \         P                  P                  4       '       d!   \         P                  P                  4        R # R # r   )r+   r&   r4   Úempty_cacher'   )r   rD   r   s   &&,r   Ú#_process_model_after_weight_loadingÚ4Mxfp4HfQuantizer._process_model_after_weight_loading«   sM   € ä�:‰:×"Ñ"×$Ò$Ü�J‰J×"Ñ"Ö$Ü�Y‰Y×#Ñ#×%Ò%Ü�I‰I×!Ñ!Ö#ñ &r   c                ó$   <€ V ^8„  d   QhRRRS[ /# )rC   rD   r   Úuse_kernels©rH   )rI   rJ   s   "€r   rK   rL   ²   s   ø€ ÷ 
ñ 
à ð
ñ ñ
r   c                ó,  € ^RI Hp \        P                  P	                  4       ;'       g    \        P
                  ! R4      pV'       d8   VP                  R9  d'   \        P                  R4       RV P                  n
        V'       g8   VP                  R9   d'   \        P                  R4       RV P                  n
        V P                  WP                  P                  VP                  4      V n        V! WP                  V P                  R7      pR# )	rC   )Úreplace_with_mxfp4_linearr#   zžYou are using full precision kernels, we will dequantize the model to bf16. To use the quantized model with quantization kernels, please set use_kernels=FalseTz¯MXFP4 inference on CPU requires use_kernels=True, but use_kernels is disabled. We will dequantize the model to bf16. To run MXFP4 natively on CPU, please set use_kernels=True.)Úmodules_to_not_convertr   N)r#   )rR   ra   r+   r,   r-   r.   r/   r1   r2   r   r*   Úget_modules_to_not_convertrb   Ú_keep_in_fp32_modules)r   rD   r^   r   ra   r.   s   &&&,  r   Ú$_process_model_before_weight_loadingÚ5Mxfp4HfQuantizer._process_model_before_weight_loading²   sÚ   € õ 	=ô ×"Ñ"×6Ñ6Ó8×OÐO¼E¿LºLÈÓ<Oˆß˜6Ÿ;™;¨gÔ5Ü×Ñðeôð 37ˆD×$Ñ$Ô/ç˜vŸ{™{¨gÔ5Ü×Ñðsôð 37ˆD×$Ñ$Ô/à&*×&EÑ&EØ×+Ñ+×BÑBÀE×D_ÑD_ó'
ˆÔ#ñ *Ø×*EÑ*EÐ[_×[sÑ[sô
Šr   c           
     ó    € R VP                   P                  9   d3   \        VRR4      e$   VP                  P	                  RRRRRRRR/4       V# )ÚGptOssConfigÚbase_model_tp_planNú(layers.*.mlp.experts.gate_up_proj_blocksÚgrouped_gemmú(layers.*.mlp.experts.gate_up_proj_scalesú%layers.*.mlp.experts.down_proj_blocksú%layers.*.mlp.experts.down_proj_scales)r   Ú__name__Úgetattrri   Úupdate©r   Úconfigs   &&r   Úupdate_tp_planÚMxfp4HfQuantizer.update_tp_planÓ   óZ   € Ø˜V×-Ñ-×6Ñ6Ô6Ü�vÐ3°TÓ:ÒFØ×)Ñ)×0Ñ0àBÀNØBÀNØ?ÀØ?Àð	ôð ˆr   c           
     ó    € R VP                   P                  9   d3   \        VRR4      e$   VP                  P	                  RRRRRRRR/4       V# )rh   Úbase_model_ep_planNrj   rk   rl   rm   rn   )r   ro   rp   rx   rq   rr   s   &&r   Úupdate_ep_planÚMxfp4HfQuantizer.update_ep_planà   rv   r   c                óR  € ^RI Hp VP                  4       p\        VP                  R^ 4      p\        VP                  RR4      pVP                  4        EFJ  w  rg\        Wr4      '       d%   \        VR4      '       d   \        VR4      '       g   K=  R EF  p\        Wx4      p	\        Wx R24      p
V	P                  P                  P                  V	P                  P                  4      P                  RR4      pVR8X  d   VP                  VR^Z^4      pMVP                  WE^ZR4      pV
P                  P                  P                  P                  V
P                  P                  P                  4      P                  RR4      pW³V RV R	2&   WÃV RV R
2&   EK	  	  EKM  	  / pW=3# )rC   rN   Únum_local_expertsÚhidden_sizei@  Úgate_up_projÚ	down_projÚ_precision_configÚ.Ú_blocksÚ_scales)r~   r   éÿÿÿÿéþÿÿÿ)rR   rO   Ú
state_dictrp   rs   Únamed_modulesr8   ÚhasattrÚstorageÚlayoutÚunswizzle_dataÚdataÚ	transposeÚreshapeÚweight_scale)r   rD   rO   r†   r|   r}   ÚnamerS   ÚprojÚtriton_tensorÚprecision_configÚblocksÚscalesÚmetadatas   &&            r   Úget_state_dict_and_metadataÚ,Mxfp4HfQuantizer.get_state_dict_and_metadataí   s‡  € Ý5à×%Ñ%Ó'ˆ
Ü# E§L¡LÐ2EÀrÓJÐÜ˜eŸl™l¨M¸4Ó@ˆà!×/Ñ/×1‰LˆDä˜6×6Ò6Ü˜F N×3Ò3Ü˜F K×0Ò0áä5�Ü '¨Ó 5�Ü#*¨6°VÐ;LÐ3MÓ#NÐ à&×.Ñ.×5Ñ5×DÑDÀ]×EZÑEZ×E_ÑE_Ó`×jÑjÐkmÐoqÓr�Ø˜>Ô)Ø#Ÿ^™^Ð,=¸rÀ2ÀrÓJ‘Fà#Ÿ^™^Ð,=ÈBÐPRÓS�Fà)×6Ñ6×>Ñ>×EÑE×TÑTØ$×1Ñ1×9Ñ9×>Ñ>óç‘)˜B Ó#ð ð 7=˜d˜V 1 T F¨'Ð2Ñ3Ø6<˜d˜V 1 T F¨'Ð2Ô3ô 6ñ 2ð2 ˆØÐ#Ð#r   c                ó   € R # )TrX   ©r   s   &r   Úis_serializableÚ Mxfp4HfQuantizer.is_serializable  s   € Ùr   c                ó    <€ V ^8„  d   QhRS[ /# )rC   rF   r_   )rI   rJ   s   "€r   rK   rL     s   ø€ ÷ ñ ™dñ r   c                ó0   € \         P                  R 4       R# )z©MXFP4 quantization don't support training, please consider dequantizing the model first by passing quantization_config=Mxfp4Config(dequantize=True) to .from_pretrained()F)r1   r2   rš   s   &r   Úis_trainableÚMxfp4HfQuantizer.is_trainable  s   € ä×Ñð xô	
ñ r   c                ó   € ^RI Hp V! V 4      # )rC   )ÚMxfp4Quantize)Úintegrations.mxfp4r¢   )r   r¢   s   & r   Úget_quantize_opsÚ!Mxfp4HfQuantizer.get_quantize_ops  s   € Ý6á˜TÓ"Ð"r   c                ó(  € ^RI HpHp V P                  '       dL   V P                  P
                  '       d0   \        RR.RV! V 4      .R7      \        RR.R.V! V 4      .R7      .# \        RR.RV! V 4      .R7      \        RR.RV! V 4      .R7      .# )	rC   )ÚMxfp4DequantizeÚMxfp4DeserializeÚdown_proj_blocksÚdown_proj_scalesz
down_proj$)Úsource_patternsÚtarget_patternsÚ
operationsÚgate_up_proj_blocksÚgate_up_proj_scaleszgate_up_proj$)r£   r§   r¨   r0   r   r*   r   )r   r§   r¨   s   &  r   Úget_weight_conversionsÚ'Mxfp4HfQuantizer.get_weight_conversions  s»   € ßJà××Ð $×":Ñ":×"E×"EÐ"EäØ%7Ð9KÐ$LØ$1Ù /°Ó 5Ð6ôô
  Ø%:Ð<QÐ$RØ%4Ð$5Ù /°Ó 5Ð6ôðð ô Ø!6Ð8MÐ NØ 0Ù,¨TÓ2Ð3ôô
 Ø!3Ð5GÐ HØ -Ù,¨TÓ2Ð3ôð
ð 	
r   c                ó$   <€ V ^8„  d   Qh/ R;R&   # )rC   r   r   rX   )rI   rJ   s   "€r   rK   rL   *   s   ø‡ ‚ ð 'Ñ&ò r   )rb   r   )F)ro   Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__Úrequires_calibrationr   r    r@   rU   r[   re   rt   ry   r—   r›   ÚpropertyrŸ   r¤   r°   Ú__annotate_func__Ú__static_attributes__Ú__classdictcell__Ú__classcell__)r   rJ   s   @@r   r   r   *   s}   ù‡ € ñð !Ðõ'ò	'ò^÷@ð ÷$ð $÷
ò 
òBòò!$òFð ÷ó ðò#ò

÷k … r   r   )Útypingr   Úbaser   Úmodeling_utilsr   Úutils.quantization_configr   Úutilsr   r   r	   r
   r   Úquantizers_utilsr   r+   Úcore_model_loadingr   Ú
get_loggerro   r1   r   r   rX   r   r   Ú<module>rÅ      s\   ðõ !å ÷ Ý0Ý7÷õ õ 3ñ ×ÒÛå4à	×	Ò	˜HÓ	%€ØÐ ôQ
�{ö Q
r   