+
    G-jM+  ã                   ó¦   € ^ RI t ^ RIt^ RIt^ RIt^ RIt^ RIt^ RIHtH	t	H
t
Ht ^ RIHtHt  ! R R]4      tR tR tR t]R	8X  d
   ]! 4        R# R# )
é    N)ÚQuantFormatÚ	QuantTypeÚStaticQuantConfigÚquantize)ÚCalibrationDataReaderÚCalibrationMethodc                   ó>   a € ] tR t^t o R tV 3R lR ltR tRtV tR# )ÚOnnxModelCalibrationDataReaderc           
     óx  € \         P                  P                  V4      V n        \         P                  ! V P                  4       Uu. uFE  q"P                  R 4      '       g   K  \         P                  P                  V P                  V4      NKG  	  pp\        P                  ! V4      P                  4       p. pV Fž  p/ p\        \        V4      4       Uu. uF'  p\         P                  P                  VRV R24      NK)  	  p	pV	 U
u. uF  q P                  V
4      NK  	  pp
\        WKRR7       F  w  rÍW×VP                  &   K  	  VP                  V4       K   	  \        V4      \        V4      8X  g   Q h\        V^ ,          4      \        V4      8X  g   Q h\!        V4      V n        R# u upi u upi u up
i )Útest_data_set_Úinput_z.pbF)ÚstrictN)ÚosÚpathÚdirnameÚ	model_dirÚlistdirÚ
startswithÚjoinÚonnxruntimeÚInferenceSessionÚ
get_inputsÚrangeÚlenÚread_onnx_pb_dataÚzipÚnameÚappendÚiterÚcalibration_data)ÚselfÚ
model_pathÚaÚ	data_dirsÚmodel_inputsÚname2tensorsÚdata_dirÚname2tensorÚ	input_idxÚ
data_pathsÚ	data_pathÚdata_ndarraysÚmodel_inputÚdata_ndarrays   &&            Ú€/Volumes/fast/ai/experiments/nudenet-smoke/.venv/lib/python3.14/site-packages/onnxruntime/quantization/static_quantize_runner.pyÚ__init__Ú'OnnxModelCalibrationDataReader.__init__   sl  € ÜŸ™Ÿ™¨Ó4ˆŒä57·Z²ZÀÇÁÔ5Oó
Ù5O°×S_ÑS_Ð`p×SqÔ+ŒB�G‰G�L‰L˜Ÿ™¨Ö+Ñ5Oð 	ð 
ô #×3Ò3°JÓ?×JÑJÓLˆØˆÛ!ˆHØˆKÜ[`ÔadÐeqÓarÔ[sÓtÑ[sÈiœ"Ÿ'™'Ÿ,™, x°6¸)¸ÀCÐ1HÖIÑ[sˆJÐtÙPZÓ[ÑPZÀ9×3Ñ3°IÖ>ÑPZˆMÐ[Ü-0°ÐUZ×-[Ñ)�Ø0<˜K×,Ñ,Ó-ñ .\à×Ñ Ö,ñ "ô �<Ó ¤C¨	£NÔ2Ð2Ð2Ü�< •?Ó#¤s¨<Ó'8Ô8Ð8Ð8ä $ \Ó 2ˆÖùò
ùò uùÚ[s   ÁF-Á$.F-Ã-F2Ä
F7c                ó    <€ V ^8„  d   QhRS[ /# )é   Úreturn)Údict)ÚformatÚ__classdict__s   "€r/   Ú__annotate__Ú+OnnxModelCalibrationDataReader.__annotate__!   s   ø€ ÷ 1ñ 1™$ñ 1ó    c                ó.   € \        V P                  R4      # )z9generate the input data dict for ONNXinferenceSession runN)Únextr    )r!   s   &r/   Úget_nextÚ'OnnxModelCalibrationDataReader.get_next!   s   € ä�D×)Ñ)¨4Ó0Ð0r:   c                ó  € \         P                  ! 4       p\        VR 4      ;_uu_ 4       pVP                  VP	                  4       4       RRR4       \         P
                  P                  V4      pV#   + '       g   i     L1; i)ÚrbN)ÚonnxÚTensorProtoÚopenÚParseFromStringÚreadÚnumpy_helperÚto_array)r!   Úfile_pbÚtensorÚfÚrets   &&   r/   r   Ú0OnnxModelCalibrationDataReader.read_onnx_pb_data%   s\   € Ü×!Ò!Ó#ˆÜ�'˜4× Ô  AØ×"Ñ" 1§6¡6£8Ô,÷ !ä×Ñ×(Ñ(¨Ó0ˆØˆ
÷ !× ús   ª A3Á3B	)r    r   N)	Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__r0   r=   r   Ú__static_attributes__Ú__classdictcell__)r7   s   @r/   r
   r
      s   ø‡ € ò3÷&1ð 1÷ð r:   r
   c            	      ó@  € \         P                  ! R R7      p V P                  RRRRR7       V P                  RRRR	R7       V P                  R
. RGORRR7       V P                  R. RGORRR7       V P                  RRRR7       V P                  RRRR7       V P                  RRRR7       V P                  RRRR7       V P                  RRRR7       V P                  RR. RR 7       V P                  R!R". RHOR#R$7       V P                  R%R&R&R'.R(R$7       V P                  R)RR*R7       V P                  R+RR,R7       V P                  R-RR.R7       V P                  R/RR0R7       V P                  R1\        R2R3R47       V P                  R5RR6R7       V P                  R7RR8R7       V P                  R9RR:R7       V P                  R;RR<R=R 7       V P                  R>RR<R?R 7       V P                  R@^RARI. RBRC7       V P                  RDRERF7       V P	                  4       # )Jz%The arguments for static quantization)Údescriptionz-iz--input_model_pathTzPath to the input onnx model)ÚrequiredÚhelpz-oz--output_quantized_model_pathz'Path to the output quantized onnx modelz--activation_typeÚqint8Úquint8z!Activation quantization type used)ÚchoicesÚdefaultrV   z--weight_typezWeight quantization type usedz--enable_subgraphÚ
store_truez#If set, subgraph will be quantized.)ÚactionrV   z--force_quantize_no_input_checka   By default, some latent operators like maxpool, transpose, do not quantize if their input is not quantized already. Setting to True to force such operator always quantize input and so generate quantized output. Also the True behavior could be disabled per node using the nodes_to_exclude.z--matmul_const_b_onlyz3If set, only MatMul with const B will be quantized.z--add_qdq_pair_to_weightzjIf set, it remains floating-point weight and inserts both QuantizeLinear/DeQuantizeLinear nodes to weight.z--dedicated_qdq_pairzFIf set, it will create identical and dedicated QDQ pair for each node.z)--op_types_to_exclude_output_quantizationÚ+z]If any op type is specified, it won't quantize the output of ops with this specific op types.)ÚnargsrZ   rV   z--calibration_methodÚminmaxzCalibration method used)rZ   rY   rV   z--quant_formatÚqdqÚ	qoperatorzQuantization format usedz--calib_tensor_range_symmetriczoIf enabled, the final range of tensor during calibration will be explicitly set to symmetric to central point 0z--calib_moving_averagez�If enabled, the moving average of the minimum and maximum values will be computed when the calibration method selected is MinMax.z--disable_quantize_biaszÃWhether to quantize floating-point biases by solely inserting a DeQuantizeLinear node If not set, it remains floating-point bias and does not insert any quantization nodes associated with biases.z--use_qdq_contrib_opszÃIf set, the inserted QuantizeLinear and DequantizeLinear ops will have the com.microsoft domain, which forces use of ONNX Runtime's QuantizeLinear and DequantizeLinear contrib op implementations.z--minimum_real_rangeg-Cëâ6?a�  If set to a floating-point value, the calculation of the quantization parameters (i.e., scale and zero point) will enforce a minimum range between rmin and rmax. If (rmax-rmin) is less than the specified minimum range, rmax will be set to rmin + MinimumRealRange. This is necessary for EPs like QNN that require a minimum floating-point range when determining  quantization parameters.)ÚtyperZ   rV   z --qdq_keep_removable_activationsz|If set, removable activations (e.g., Clip or Relu) will not be removed, and will be explicitly represented in the QDQ model.z*--qdq_disable_weight_adjust_for_int32_biasz‚If set, QDQ quantizer will not adjust the weight's scale when the bias has a scale (input_scale * weight_scale) that is too small.z--per_channelz&Whether using per-channel quantizationz--nodes_to_quantizeNzfList of nodes names to quantize. When this list is not None only the nodes in this list are quantized.z--nodes_to_excludeznList of nodes names to exclude. The nodes in this list will be excluded from quantization when it is not None.z--op_per_channel_axisr   a8  Set channel axis for specific op type, for example: --op_per_channel_axis MatMul 1, and it's effective only when per channel quantization is supported and per_channel is True. If specific op type supports per channel quantization but not explicitly specified with channel axis, default channel axis will be used.)r^   r\   ÚmetavarrZ   rV   z--tensor_quant_overridesz4Set the json file for tensor quantization overrides.)rV   )rW   rX   Úqint16Úquint16Úqint4Úquint4Úqfloat8e4m3fn)r_   ÚentropyÚ
percentileÚdistribution)ÚOP_TYPEÚPER_CHANNEL_AXIS)ÚargparseÚArgumentParserÚadd_argumentÚfloatÚ
parse_args)Úparsers    r/   Úparse_argumentsrt   -   s  € Ü×$Ò$Ð1XÔY€FØ
×Ñ˜Ð2¸TÐHfÐÔgØ
×ÑØÐ-¸ÐClð ô ð ×ÑØÚ\ØØ0ð	 ô ð ×ÑØÚ\ØØ,ð	 ô ð ×ÑÐ+°LÐGlÐÔmØ
×ÑØ)Øðkð ô ð ×ÑØØØBð ô ð
 ×ÑØ"Øðð ô ð ×ÑØØØUð ô ð
 ×ÑØ3ØØØlð	 ô ð ×ÑØØÚCØ&ð	 ô ð ×ÑÐ(°%À%ÈÐAUÐ\vÐÔwØ
×ÑØ(Øð/ð ô ð ×ÑØ Øðkð ô ð ×ÑØ!Øð#ð ô ð ×ÑØØðnð ô ð ×ÑØÜØð$ð	 ô 	ð ×ÑØ*Øð@ð ô ð ×ÑØ4ØðGð ô ð ×Ñ˜°ÐCkÐÔlØ
×ÑØØØØuð	 ô ð ×ÑØØØØ}ð	 ô ð ×ÑØØØØ/Øð.ð ô 
ð ×ÑÐ2Ð9oÐÔpØ×ÑÓÐr:   c                 ót  € V '       g   / # \        V 4      ;_uu_ 4       p\        P                  ! V4      pR R R 4       X Fb  pW#,           FS  p\        P                  ! VR,          \        P
                  R7      VR&   \        P                  ! VR,          4      VR&   KU  	  Kd  	  V#   + '       g   i     Lz; i)NÚscale)ÚdtypeÚ
zero_point)rC   ÚjsonÚloadÚnpÚarrayÚfloat32)ÚfilerJ   Úquant_override_dictrI   Úenc_dicts   &    r/   Úget_tensor_quant_overridesr�   µ   sˆ   € çØˆ	Ü	ˆd�Œ�qÜ"Ÿiši¨›lÐ÷ 
ã%ˆØ+×3Ð3ˆHÜ "§¢¨°'Õ):Ä"Ç*Á*Ô MˆH�WÑÜ%'§X¢X¨h°|Õ.DÓ%EˆH�\Ó"ó 4ñ &ð Ð÷ 
�ús   žB'Â'B7	c                   óÈ  € \        4       p \        V P                  R 7      pR\        P                  R\        P
                  R\        P                  R\        P                  R\        P                  R\        P                  R\        P                  /pW P                  ,          pW P                  ,          p\        V P                  4      pRV P                  R	V P                   R
V P"                  RV P$                  RV P&                  RV P(                  RVRV P*                  RV P,                  RV P.                  '       * RV P0                  RV P2                  RV P4                  RV P6                  R\9        V P:                  4      /pR\<        P>                  R\<        P@                  R\<        PB                  R\<        PD                  /pR\F        PH                  R\F        PJ                  /p\M        VWpPN                  ,          W€PP                  ,          VVRV PR                  V PT                  V PV                  RRRVR7      p	\Y        V P                  V PZ                  V	R 7       R# )!)r"   rW   rX   rd   re   rf   rg   rh   ÚEnableSubgraphÚForceQuantizeNoInputCheckÚMatMulConstBOnlyÚAddQDQPairToWeightÚ"OpTypesToExcludeOutputQuantizationÚDedicatedQDQPairÚ QDQOpTypePerChannelSupportToAxisÚCalibTensorRangeSymmetricÚCalibMovingAverageÚQuantizeBiasÚUseQDQContribOpsÚMinimumRealRangeÚQDQKeepRemovableActivationsÚ"QDQDisableWeightAdjustForInt32BiasÚTensorQuantOverridesr_   ri   rj   rk   r`   ra   NF)Úcalibration_data_readerÚcalibrate_methodÚquant_formatÚactivation_typeÚweight_typeÚop_types_to_quantizeÚnodes_to_quantizeÚnodes_to_excludeÚper_channelÚreduce_rangeÚuse_external_data_formatÚcalibration_providersÚextra_options)r-   Úmodel_outputÚquant_config).rt   r
   Úinput_model_pathr   ÚQInt8ÚQUInt8ÚQInt16ÚQUInt16ÚQInt4ÚQUInt4ÚQFLOAT8E4M3FNr•   r–   r5   Úop_per_channel_axisÚenable_subgraphÚforce_quantize_no_input_checkÚmatmul_const_b_onlyÚadd_qdq_pair_to_weightÚ'op_types_to_exclude_output_quantizationÚdedicated_qdq_pairÚcalib_tensor_range_symmetricÚcalib_moving_averageÚdisable_quantize_biasÚuse_qdq_contrib_opsÚminimum_real_rangeÚqdq_keep_removable_activationsÚ(qdq_disable_weight_adjust_for_int32_biasr�   Útensor_quant_overridesr   ÚMinMaxÚEntropyÚ
PercentileÚDistributionr   ÚQDQÚ	QOperatorr   Úcalibration_methodr”   r˜   r™   rš   r   Úoutput_quantized_model_path)
ÚargsÚdata_readerÚarg2quant_typer•   r–   Ú'qdq_op_type_per_channel_support_to_axisrž   Úarg2calib_methodÚarg2quant_formatÚsqcs
             r/   ÚmainrÇ   Â   s  € ÜÓ€DÜ0¸D×<QÑ<QÔR€Kà”—‘Ø”)×"Ñ"Ø”)×"Ñ"Ø”9×$Ñ$Ø”—‘Ø”)×"Ñ"Øœ×0Ñ0ð€Nð %×%9Ñ%9Õ:€OØ ×!1Ñ!1Õ2€KÜ.2°4×3KÑ3KÓ.LÐ+à˜$×.Ñ.Ø# T×%GÑ%GØ˜D×4Ñ4Ø˜d×9Ñ9Ø,¨d×.ZÑ.ZØ˜D×3Ñ3Ø*Ð,SØ# T×%FÑ%FØ˜d×7Ñ7Ø˜D×6Ñ6Ô6Ø˜D×4Ñ4Ø˜D×3Ñ3Ø% t×'JÑ'JØ,¨d×.[Ñ.[àÔ :¸4×;VÑ;VÓ Wð!€Mð& 	Ô#×*Ñ*ØÔ$×,Ñ,ØÔ'×2Ñ2ØÔ)×6Ñ6ð	Ðð 	Œ{�‰Ø”[×*Ñ*ðÐô Ø +Ø)×*AÑ*AÕBØ%×&7Ñ&7Õ8Ø'ØØ!Ø×0Ñ0Ø×.Ñ.Ø×$Ñ$ØØ!&Ø"Ø#ô€Cô ˜×.Ñ.¸T×=]Ñ=]Ðlo×pr:   Ú__main__)rn   ry   r   Únumpyr{   rA   r   Úonnxruntime.quantizationr   r   r   r   Ú"onnxruntime.quantization.calibrater   r   r
   rt   r�   rÇ   rM   © r:   r/   Ú<module>rÍ      sU   ðÛ Û Û 	ã Û ã ß XÓ Xß WôÐ%:ô ò@EòP
ò:qðz ˆzÔÙ†Fñ r:   