+
    QV-j.%  ã                  óÊ   € ^ RI Ht ^ RIHt ^RIHtHt ^RIHt ^RI	H
t
 ^RIHt ]! 4       '       d   ^ RIt]'       d   ^RIHt ]P                   ! ]4      t ! R	 R
]
4      tR# )é    )Úannotations)ÚTYPE_CHECKING)Úis_torch_availableÚlogging)Ú
SinqConfig)ÚHfQuantizer)Úget_module_from_nameN)ÚPreTrainedModelc                  óÌ   a € ] tR t^!t$ RtRtR]R&   R]R&   R V 3R lltR	 R
 lt]	R R l4       t
R tR R ltR R ltR R ltR R ltR tR tRR R lltR R ltRtV ;t# )ÚSinqHfQuantizera|  
HF v5 quantizer for SINQ.

Modes:
  - method="sinq" (default):
      * weight-only SINQ
      * param-level ConversionOps (`SinqQuantize`) during load for pure language models
        (each Linear.weight is turned into a SINQLinear module)
      * module-level quantization after load for multimodal models
  - method="asinq":
      * A-SINQ (activation-aware) SINQ quantization
TÚboolÚ requires_parameters_quantizationr   Úquantization_configc               ó   € V ^8„  d   QhRR/# )é   r   r   © )Úformats   "Úw/Volumes/fast/ai/experiments/ui-tars-smoke/.venv/lib/python3.14/site-packages/transformers/quantizers/quantizer_sinq.pyÚ__annotate__ÚSinqHfQuantizer.__annotate__2   s   € ÷ 0ñ 0¨Jñ 0ó    c                	óF   <€ \         SV `  ! V3/ VB  R V n        RV n        R # )NF)ÚsuperÚ__init__Ú_normalized_device_strÚ_do_param_level_sinq)Úselfr   ÚkwargsÚ	__class__s   &&,€r   r   ÚSinqHfQuantizer.__init__2   s&   ø€ Ü‰ÒÐ,Ñ7°Ò7à26ˆÔ#Ø*/ˆÖ!r   c               ó   € V ^8„  d   QhRR/# ©r   Úreturnr   r   )r   s   "r   r   r   8   s   € ÷ ñ  ñ r   c                	ó   € R # ©Tr   ©r   s   &r   Úis_serializableÚSinqHfQuantizer.is_serializable8   s   € Ùr   c               ó   € V ^8„  d   QhRR/# r"   r   )r   s   "r   r   r   <   s   € ÷ ñ ˜dñ r   c                	ó   € R # r%   r   r&   s   &r   Úis_trainableÚSinqHfQuantizer.is_trainable;   s   € ár   c                	óÒ   € Vfc   \         P                  P                  4       '       d"   R\         P                  P                  4       /pMRR/p\        P                  RV R24       V# )NÚ Úcpuz:The device_map was not initialized. Setting device_map to zJ. If you want to use the model for inference, please set device_map='auto')ÚtorchÚcudaÚis_availableÚcurrent_deviceÚloggerÚinfo)r   Ú
device_maps   &&r   Úupdate_device_mapÚ!SinqHfQuantizer.update_device_map?   se   € ØÒÜ�z‰z×&Ñ&×(Ò(Ø ¤%§*¡*×";Ñ";Ó"=Ð>‘
à  %˜[�
Ü�K‰Kð)Ø)3¨ð 5[ð[ôð
 Ðr   c               ó    € V ^8„  d   QhRRRR/# )r   Údtypeztorch.dtyper#   r   )r   s   "r   r   r   L   s   € ÷ ñ  +ð °+ñ r   c                	ó:   € Vf   \         P                  pWn        V# ©N)r0   Úbfloat16r:   )r   r:   s   &&r   Úupdate_dtypeÚSinqHfQuantizer.update_dtypeL   s   € ØŠ=Ü—N‘NˆEØŒ
Øˆr   c               ó   € V ^8„  d   QhRR/# )r   r#   ÚNoner   )r   s   "r   r   r   R   s   € ÷ ñ °tñ r   c                	óø  € ^RI Hp V! 4       '       g   \        R4      h\        P                  P                  4       '       g   \        P                  R4       VP                  R4      p\        V\        4      '       dB   \        VP                  4       4      p\        V4      ^8”  d   \        R\        V4       R24      hV P                   P"                  R8X  d    V P$                  '       g   \'        R4      hR	# R	# )
r   )Úis_sinq_availablezMThe 'sinq' package is not installed. Please install it with: pip install sinqz¯No CUDA device is available. Quantization and inference will run on the CPU. Please note that this will significantly slow down inference speed and increase quantization time.r6   zkSinqHfQuantizer: multi-GPU device_map detected, but SINQ currently supports only a single CUDA device. Got z. Please use device_map=None.ÚasinqzßYou are using `method='asinq'` in the quantization config. Right now the calibrated version of SINQ is not supported in Hugging Face, please refer and use the official SINQ repository `to quantize a model with this method. N)ÚutilsrC   ÚImportErrorr0   r1   r2   r4   ÚwarningÚgetÚ
isinstanceÚdictÚsetÚvaluesÚlenÚRuntimeErrorÚsortedr   ÚmethodÚpre_quantizedÚ
ValueError)r   Úargsr   rC   r6   Údevice_map_valuess   &*,   r   Úvalidate_environmentÚ$SinqHfQuantizer.validate_environmentR   sá   € Ý-á ×"Ò"ÜÐmÓnÐnä�z‰z×&Ñ&×(Ò(Ü�N‰Nð Bôð —Z‘Z Ó-ˆ
ä�j¤$×'Ò'Ü # J×$5Ñ$5Ó$7Ó 8ÐÜÐ$Ó%¨Ô)Ü"ð#Ü#)Ð*;Ó#<Ð"=Ð=Zð\óð ð
 ×#Ñ#×*Ñ*¨gÔ5¸d×>P×>PÐ>PÜð:óð ñ ?QÑ5r   c               ó    € V ^8„  d   QhRRRR/# )r   Úcfgr   r#   rJ   r   )r   s   "r   r   r   n   s   € ÷ 
ñ 
¨*ð 
¸ñ 
r   c                óØ   € ^ RI Hp VP                  pT! \        VP                  4      VP
                  e   \        VP
                  4      MRRRR^\        VP                  4      VR7      # )z9
Build the dict that SINQLinear expects as quant_config.
)Úsinq_base_quant_configNF)ÚnbitsÚ
group_sizeÚ
quant_zeroÚquant_scaleÚview_as_floatÚaxisÚtiling_moderP   )Úsinq.sinqlinear_hfrZ   rP   Úintr[   r\   Ústrra   )r   rX   Úsinq_base_quant_config_fnrP   s   &&  r   Ú_build_sinq_quant_dictÚ&SinqHfQuantizer._build_sinq_quant_dictn   s[   € õ 	[à—‘ˆÙ(Ü�c—i‘i“.Ø.1¯n©nÒ.H”s˜3Ÿ>™>Ô*ÈdØØØØÜ˜CŸO™OÓ,Øô	
ð 		
r   c               ó$   € V ^8„  d   QhRRRRRR/# )r   Úmodelr
   Ú
param_namerd   r#   r   r   )r   s   "r   r   r   €   s"   € ÷  ñ  ¨oð  È3ð  Ð]añ  r   c                ó  € ^ RI Hp V P                  '       d   R# V P                  P                  R8X  d   R# V P
                  '       g   R# \        W4      w  rVVR8w  d   R# \        WT4      p\        VRR4      pT;'       d    V'       * p	V	# )aõ  
Called per-parameter to decide whether to run `SinqQuantize` on it.

- If `self.pre_quantized`, we do *not* quantize again (handled by SinqDeserialize instead).
- For method="asinq": return False (ASINQ is not supported in Hugging Face).
- For method="sinq": True only for SINQLinear.weight not in modules_to_not_convert.

Note: After _process_model_before_weight_loading(), the modules are already SINQLinear,
not nn.Linear. We check for SINQLinear modules that are not yet quantized (ready=False).
)Ú
SINQLinearFrD   ÚweightÚreadyT)	rb   rl   rQ   r   rP   r   r	   rI   Úgetattr)
r   ri   rj   r   rl   ÚmoduleÚtensor_nameÚis_sinqÚis_readyÚresults
   &&&,      r   Úparam_needs_quantizationÚ(SinqHfQuantizer.param_needs_quantization€   s„   € õ 	2à××ÐÙà×#Ñ#×*Ñ*¨gÔ5Ùð ×(×(Ð(Ùä2°5ÓEÑˆà˜(Ô"Ùô ˜VÓ0ˆÜ˜6 7¨DÓ1ˆØ×)Ð) œ\ˆØˆr   c                ó   € ^RI Hp V! V 4      # )zƒ
Return the ConversionOps used for param-level quantization (Sinq).
The actual SINQLinear construction is in integrations/sinq.py.
)ÚSinqQuantize)Úintegrations.sinqrx   )r   rx   s   & r   Úget_quantize_opsÚ SinqHfQuantizer.get_quantize_ops¢   s   € õ
 	5á˜DÓ!Ð!r   c                ón   € ^RI Hp V P                  '       d   ^RIHp V! . ROR.V! V 4      .R7      .# . # )zü
If `pre_quantized=True`, interpret a checkpoint produced by SINQLinear.state_dict:

    <prefix>.W_q
    <prefix>.bias
    <prefix>.meta

via a WeightConverter + SinqDeserialize so that we reconstruct a SINQLinear
module instead of a plain nn.Linear.
)ÚWeightConverter)ÚSinqDeserializez.weight)Úsource_patternsÚtarget_patternsÚ
operations)z.W_qz.metaz.bias)Úcore_model_loadingr}   rQ   ry   r~   )r   r}   r~   s   &  r   Úget_weight_conversionsÚ&SinqHfQuantizer.get_weight_conversions«   sH   € õ 	9à××ÐÝ;ñ  ò%ð
 &/ KÙ /°Ó 5Ð6ôð
ð 
ð ˆ	r   c               ó    € V ^8„  d   QhRRRR/# )r   ri   r
   Úkeep_in_fp32_moduleszlist[str] | Noner   )r   s   "r   r   r   È   s   € ÷ )
ñ )
àð)
ð /ñ	)
r   c           	     ó²  € ^RI Hp T P                  YP                  P                  ;'       g    . V4      V n        V P                  P
                  R8H  ;'       d    V P                  '       * V n        V P                  '       d   RMV P                  V P                  4      p\        V\        4      '       dL   \        \        VP                  4       4      ^ 4      p\        V\        4      '       d   RV 2pM4\        V4      pM(\         P"                  P%                  4       '       d   RMRpV! VV P                  VV P&                  VV P                  R7      pR# )zä
Called on meta-initialized model, before loading any weights.

For SINQ, we replace nn.Linear modules with empty SINQLinear modules here.
The actual quantization happens later in SinqQuantize.convert() when weights are loaded.
)Úreplace_with_sinq_linearÚsinqNzcuda:zcuda:0r/   )Úmodules_to_not_convertÚquant_configÚcompute_dtypeÚdevicerQ   )ry   rˆ   Úget_modules_to_not_convertr   rŠ   rP   rQ   r   rf   rI   rJ   ÚnextÚiterrL   rc   rd   r0   r1   r2   r:   )	r   ri   r6   r†   r   rˆ   Úsinq_quant_dictÚfirst_deviceÚ
device_strs	   &&&&,    r   Ú$_process_model_before_weight_loadingÚ4SinqHfQuantizer._process_model_before_weight_loadingÈ   s  € õ 	Aà&*×&EÑ&EØ×,Ñ,×CÑC×IÐIÀrÐL`ó'
ˆÔ#ð
 %)×$<Ñ$<×$CÑ$CÀvÑ$M×$hÐ$hÐVZ×VhÑVhÔRhˆÔ!à"&×"4×"4Ð"4™$¸$×:UÑ:UÐVZ×VnÑVnÓ:oˆô �j¤$×'Ò'Ü¤ Z×%6Ñ%6Ó%8Ó 9¸1Ó=ˆLÜ˜,¬×,Ò,Ø$ \ NÐ3‘
ä  Ó.‘
ä%*§Z¡Z×%<Ñ%<×%>Ò%>™ÀEˆJá(ØØ#'×#>Ñ#>Ø(ØŸ*™*ØØ×,Ñ,ô
Šr   c               ó   € V ^8„  d   QhRR/# )r   ri   r
   r   )r   s   "r   r   r   ó   s   € ÷ ñ àñr   c                ó    € ^ RI Hp V! 4        V# )a9  
Called after *all* weights have been loaded.

For SINQ:
1. Move non-SINQLinear modules to GPU (embeddings, norms, lm_head, etc.)
   - SINQLinear modules already have GemLite buffers on GPU
   - We skip moving SINQLinear's W_q/meta to avoid memory duplication
2. Patch HF save/load methods for SINQ serialization
)Úpatch_hf_pretrained_io)Ú
sinq.hf_ior˜   )r   ri   r   r˜   s   &&, r   Ú#_process_model_after_weight_loadingÚ3SinqHfQuantizer._process_model_after_weight_loadingó   s   € õ 	6ñ 	Ô àˆr   )r   r   r:   rŠ   r<   )Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__r   Ú__annotations__r   r'   Úpropertyr+   r7   r>   rU   rf   ru   rz   rƒ   r”   rš   Ú__static_attributes__Ú__classcell__)r   s   @r   r   r   !   sr   ø‡ ñð .2Ð$ dÓ1Ø#Ó#÷0ð 0õð ôó ðòõõõ8
õ$ òD"ò÷:)
÷Vó r   r   )Ú
__future__r   Útypingr   rE   r   r   Úutils.quantization_configr   Úbaser   Úquantizers_utilsr	   r0   Úmodeling_utilsr
   Ú
get_loggerrœ   r4   r   r   r   r   Ú<module>r¬      sK   ðõ #å  ç /Ý 2Ý Ý 2ñ ×ÒÛçÝ0à	×	Ò	˜HÓ	%€ôe�kö er   