Ë
    ÝÍ:j( ã                  óä  — d dl mZ d dlZd dlmZ d dlmZ d dlmZ d dl	Z
d dlZd dlmZ d dlmZ dd	lmZmZ dd
lmZ ddlmZmZmZmZmZmZmZmZmZmZmZm Z m!Z!m"Z"m#Z#m$Z$m%Z%m&Z&m'Z'm(Z(m)Z)m*Z*m+Z+ ddl,m-Z-  G d„ de«      Z.e G d„ d«      «       Z/ G d„ d«      Z0e G d„ d«      «       Z1e G d„ d«      «       Z2e G d„ d«      «       Z3e G d„ d«      «       Z4e G d„ d«      «       Z5 G d„ de«      Z6y)é    )ÚannotationsN)Ú	dataclass)ÚEnum)ÚAny)ÚTensorProto)Úonnx_pbé   )ÚBaseQuantizerÚQuantizationParams)Ú
TensorData)ÚDEQUANT_OP_NAMEÚONNX_TYPE_TO_NP_TYPEÚQUANT_OP_NAMEÚQuantizedValueÚQuantizedValueTypeÚ__producer__Ú__version__Úadd_dequant_output_suffixÚadd_dequant_suffixÚadd_quant_input_suffixÚadd_quant_output_suffixÚadd_quant_suffixÚcompute_data_quant_paramsÚcompute_scale_zpÚcompute_scale_zp_blockedÚcompute_scale_zp_float8Úfind_by_nameÚget_qmin_qmax_for_qTypeÚ	ms_domainÚnormalize_axisÚquantize_onnx_initializerÚsnap_zero_point_to_uint8Útensor_proto_to_array)ÚCreateQDQQuantizerc                  ó   — e Zd ZdZdZdZy)ÚQDQQuantTensorTyper   r	   é   N)Ú__name__Ú
__module__Ú__qualname__Ú
ACTIVATIONÚWEIGHTÚBIAS© ó    ú{/home/mcse/projects/srt_converter/srt-converter-venv/lib/python3.12/site-packages/onnxruntime/quantization/qdq_quantizer.pyr&   r&   0   s   „ Ø€JØ€FØ�Dr/   r&   c                  ó"   — e Zd ZU ded<   ded<   y)ÚQDQQuantParamProviderÚstrÚ
input_nameÚ	node_nameN©r(   r)   r*   Ú__annotations__r.   r/   r0   r2   r2   9   s   … àƒOØ„Nr/   r2   c                  ó0   — e Zd Zej                  dddfd„Zy)ÚQDQTensorQuantInfoNc                óV   — || _         || _        || _        |d u| _        |€J ‚|| _        y ©N)Útensor_typeÚquant_para_providerÚaxisÚ	is_sharedÚ	data_type)Úselfr<   r=   r>   r@   s        r0   Ú__init__zQDQTensorQuantInfo.__init__B   s8   € Ø&ˆÔØ#6ˆÔ ØˆŒ	Ø,°DÐ8ˆŒØÐ$Ð$Ð$Ø"ˆ�r/   )r(   r)   r*   r&   r+   rB   r.   r/   r0   r9   r9   A   s   „ Ø#5×#@Ñ#@ÐVZÐaeÐquô #r/   r9   c                  ó6   — e Zd ZU ded<   ded<   ded<   ded<   y)ÚQDQBiasQuantInfor3   r5   r4   Úweight_nameÚfloatÚbetaNr6   r.   r/   r0   rD   rD   L   s   … àƒNØƒOØÓØ
„Kr/   rD   c                  ó4   — e Zd ZU ded<   ded<   ded<   d	d„Zy)
ÚQDQTensorQuantParamsr   ÚoriginalzQuantizationParams | NoneÚ	convertedúset[str] | NoneÚconverted_recv_nodesc                ó®   — | j                   €| j                  S | j                  €| j                   S || j                  v r| j                   S | j                  S r;   ©rK   rJ   rM   ©rA   Úconsumer_node_names     r0   Úget_for_consumerz%QDQTensorQuantParams.get_for_consumer]   óQ   € Ø�>‰>Ð!Ø—=‘=Ð à×$Ñ$Ð,Ø—>‘>Ð!ð
 #5¸×8QÑ8QÑ"Qˆt�~‰~ÐeÐX\×XeÑXeÐer/   N)Úreturnr   ©r(   r)   r*   r7   rR   r.   r/   r0   rI   rI   W   s   … à Ó Ø(Ó(Ø)Ó)ô
fr/   rI   c                  ó"   — e Zd ZU ded<   ded<   y)ÚQDQScaleZpInitializersr   ÚscaleÚ
zero_pointNr6   r.   r/   r0   rW   rW   k   s   … àÓØÔr/   rW   c                  ó,   — e Zd ZU ded<   ded<   ded<   y)ÚQDQTensorScaleZpInitializersrW   rJ   zQDQScaleZpInitializers | NonerK   rL   rM   Nr6   r.   r/   r0   r[   r[   t   s   … à$Ó$Ø,Ó,Ø)Ô)r/   r[   c                  ó4   — e Zd ZU ded<   ded<   ded<   d	d„Zy)
ÚQDQTensorQuantizedValuer   rJ   zQuantizedValue | NonerK   rL   rM   c                ó®   — | j                   €| j                  S | j                  €| j                   S || j                  v r| j                   S | j                  S r;   rO   rP   s     r0   rR   z(QDQTensorQuantizedValue.get_for_consumer„   rS   r/   N)rT   r   rU   r.   r/   r0   r]   r]   ~   s   … àÓØ$Ó$Ø)Ó)ô
fr/   r]   c                  ó¢  — e Zd Z	 d$d„Zd„ Zd„ Zdej                  fd„Zd%d„Z	d&d„Z
d%d„Zd	„ Zd'd
„Zd(d„Z	 	 	 	 	 	 	 	 	 	 	 	 d)d„Zd„ Zd„ Zd„ Zd„ Zd„ Z	 	 d*	 	 	 	 	 	 	 	 	 	 	 	 	 d+d„Z	 	 d*	 	 	 	 	 	 	 	 	 	 	 	 	 d,d„Z	 	 d*	 d-d„Zd.d„Zd$d„Zd„ Zd„ Zd„ Zd„ Zd%d„Z	 d$	 	 	 	 	 	 	 d/d„Zd0d„Z d1d„Z!	 d2	 	 	 	 	 	 	 d3d„Z"d4d „Z#d5d!„Z$d6d"„Z%d7d#„Z&y)8ÚQDQQuantizerNc                óø  ‡— t        j                  | |||||||||	|
«       i | _        i | _        g | _        |
j                  dg «      | _        |
j                  dd«      | _        |
j                  dd«      | _        |
j                  dd«      | _	        i | _
        i | _        |
j                  di «      | _        |
j                  dd«      rt        nd | _        |
j                  d	d«      | _        |
j                  d
d«      | _        |
j                  dd«      | _        | j$                  dk  r®t&        j(                  t&        j*                  t&        j,                  t&        j.                  fŠt1        ˆfd„| j2                  D «       «      }| j                  sF| j4                  ‰v s| j6                  ‰v s|r(t9        j:                  dt        › d�«       t        | _        | j=                  «       | _        i | _         i | _!        y )NÚ"OpTypesToExcludeOutputQuantizationÚAddQDQPairToWeightFÚQuantizeBiasTÚDedicatedQDQPairÚ QDQOpTypePerChannelSupportToAxisÚUseQDQContribOpsÚQDQKeepRemovableActivationsÚ"QDQDisableWeightAdjustForInt32BiasÚ	BlockSizer   é   c              3  ó:   •K  — | ]  }|j                   ‰v –— Œ y ­wr;   )r<   )Ú.0ÚtÚopset21_typess     €r0   ú	<genexpr>z(QDQQuantizer.__init__.<locals>.<genexpr>á   s   øè ø€ ò /Ø34�—‘ Ô.ñ/ùs   ƒzÉONNX QuantizeLinear and DequantizeLinear operators do not support 16-bit/4-bit integer quantization types prior to opset 21. The domain of QuantizeLinear and DequantizeLinear operators will be set to 'z' to enable support.)"r
   rB   Útensors_to_quantizeÚbias_to_quantizeÚnodes_to_removeÚgetÚ'op_types_to_exclude_output_quantizationÚadd_qdq_pair_to_weightÚquantize_biasÚdedicated_qdq_pairÚtensor_to_its_receiving_nodesÚtensor_to_producing_dqÚ'qdq_op_type_per_channel_support_to_axisr   Úqdq_op_domainÚqdq_keep_removable_activationsÚ(qdq_disable_weight_adjust_for_int32_biasÚ
block_sizeÚopset_versionr   ÚUINT16ÚINT16ÚUINT4ÚINT4ÚanyÚtensor_quant_override_qtypesÚactivation_qTypeÚweight_qTypeÚloggingÚwarningÚcalc_graph_quant_paramsÚquantization_paramsÚinitializer_quant_paramsÚquantized_value_map)rA   ÚmodelÚper_channelÚreduce_rangerˆ   r‡   Útensors_rangeÚnodes_to_quantizeÚnodes_to_excludeÚop_types_to_quantizeÚextra_optionsÚoverrides_have_opset21_typesro   s               @r0   rB   zQDQQuantizer.__init__’   s  ø€ ô 	×ÑØØØØØØØØØØ Øô	
ð CEˆÔ Ø=?ˆÔà!ˆÔð 8E×7HÑ7HÐImÐoqÓ7rˆÔ4ð
 '4×&7Ñ&7Ð8LÈeÓ&TˆÔ#ð +×.Ñ.¨~¸tÓDˆÔð #0×"3Ñ"3Ð4FÈÓ"NˆÔØNPˆÔ*ð BDˆÔ#ð 8E×7HÑ7HÐIkÐmoÓ7pˆÔ4à*7×*;Ñ*;Ð<NÐPUÔ*V�YÐ\`ˆÔð /<×.?Ñ.?Ð@]Ð_dÓ.eˆÔ+ð 9F×8IÑ8IÐJnÐpuÓ8vˆÔ5ð  -×0Ñ0°¸aÓ@ˆŒð
 ×Ñ Ò"Ü(×/Ñ/´×1BÑ1BÄK×DUÑDUÔWb×WgÑWgÐhˆMÜ+.ó /Ø8<×8YÑ8Yô/ó ,Ð(ð ×%Ò%Ø×%Ñ%¨Ñ6Ø×$Ñ$¨Ñ5Ù/ä—‘ðcäclÐbmð n&ð&ôô &/�Ô"à#'×#?Ñ#?Ó#AˆÔ ØGIˆÔ%ð $&ˆÕ r/   c                ó  — t        || j                  j                  «       «      }|�|j                  S || j                  v rJ| j                  |   }|j
                  j                  d«      r |j
                  j                  j                  S y)ú2
        Check if tensor can be quantized
        Nr<   )	r   r�   Úinitializerr@   Úvalue_infosÚtypeÚHasFieldr<   Ú	elem_type©rA   Útensor_nameÚweightÚvis       r0   Ú_get_tensor_typezQDQQuantizer._get_tensor_type÷   sx   € ô ˜k¨4¯:©:×+AÑ+AÓ+CÓDˆØÐØ×#Ñ#Ð#Ø˜D×,Ñ,Ñ,Ø×!Ñ! +Ñ.ˆBØ�w‰w×Ñ Ô.Ø—w‘w×*Ñ*×4Ñ4Ð4Ør/   c                óú  — t        || j                  j                  «       «      }|�B|j                  t        j
                  j                  t        j
                  j                  fv ryy|| j                  v rl| j                  |   }|j                  j                  d«      rA|j                  j                  j                  t
        j                  t
        j                  fv ryyt        j                  d|› d�«       y)r™   Tr<   z$failed to infer the type of tensor: z6. Skip to quantize it. Please check if it is expected.F)r   r�   rš   r@   Ú
onnx_protor   ÚFLOATÚFLOAT16r›   rœ   r�   r<   rž   r‰   rŠ   rŸ   s       r0   Ú_is_tensor_quantizablez#QDQQuantizer._is_tensor_quantizable  sá   € ô ˜k¨4¯:©:×+AÑ+AÓ+CÓDˆØÐØ×Ñ¤J×$:Ñ$:×$@Ñ$@Ä*×BXÑBX×B`ÑB`Ð#aÑaØð ð ˜D×,Ñ,Ñ,Ø×!Ñ! +Ñ.ˆBØ�w‰w×Ñ Ô.°2·7±7×3FÑ3F×3PÑ3PÜ×!Ñ!Ü×#Ñ#ðUñ 4ð ð ô	 �O‰OØ6°{°mÐCyÐzôð r/   c                óJ  — | j                  |«      r’|rUt        |t        «      st        dt	        |«      › d�«      ‚| j                  |«      }t        |||¬«      | j                  |<   y|| j                  vr,| j                  |«      }t        ||¬«      | j                  |<   yyy)a  
        Adds a tensor to the list (actually a dict) of tensors to quantize. Called indirectly by op quantizers that
        want to quantize a tensor (i.e., "mark" a tensor for quantization).

        If quant_sharing_provider is not None, tensor with name tensor_name will be quantized with the same
        quantization parameters as the node input specified in quant_sharing_provider. Ex: A Tranpose node's output
        will typically use the same quantization parameter initializers used at the Transpose node's input.

        Args:
            tensor_name: name of the tensor to quantize
            quant_sharing_provider: name of the tensor and node that provides quantization parameter
            tensor_type: QDQQuantTensorType default ACTIVATION
        zBquant_sharing_provider must be of type QDQQuantParamProvider, not ú.)r<   r=   r@   )r<   r@   N)r¨   Ú
isinstancer2   Ú	TypeErrorrœ   r£   r9   rq   )rA   r    Úquant_sharing_providerr<   r@   s        r0   Ú__quantize_tensorzQDQQuantizer.__quantize_tensor  s¶   € ð ×&Ñ& {Ô3Ù%Ü!Ð"8Ô:OÔPÜ#Ø\Ô]aÐbxÓ]yÐ\zÐz{Ð|óð ð !×1Ñ1°+Ó>�	Ü8JØ +ÐAWÐclô9�×(Ñ(¨Ò5ð  D×$<Ñ$<Ñ<Ø ×1Ñ1°+Ó>�	Ü8JÐWbÐnwÔ8x�×(Ñ(¨Ò5ð =ð 4r/   c                óD   — | j                  |dt        j                  «      S )zó
        Adds a tensor to the list of tensors to quantize. Called by op quantizers that
        want to quantize a tensor (i.e., "mark" a tensor for quantization).

        Args:
            tensor_name: name of the tensor to quantize
        N)Ú_QDQQuantizer__quantize_tensorr&   r+   ©rA   r    s     r0   Úquantize_activation_tensorz'QDQQuantizer.quantize_activation_tensor7  s    € ð ×%Ñ% k°4Ô9K×9VÑ9VÓWÐWr/   c                óX   — | j                  |t        ||«      t        j                  «      S )a˜  
        Adds a tensor to the list of tensors to quantize. Called by op quantizers that
        want to quantize an output tensor using the same quantization parameters as one of the node's inputs.

        Ex: A Tranpose node's output will typically use the same quantization parameter initializers used at
        the Transpose node's input.

        Args:
            output_name: name of the node output to quantize so that it uses the same quantization params as an input.
            input_name: name of the node input from which the output tensor will get its quantization params.
            node_name: name of the node that consumes `input_name`.
        )r°   r2   r&   r+   )rA   Úoutput_namer4   r5   s       r0   Úquantize_output_same_as_inputz*QDQQuantizer.quantize_output_same_as_inputA  s-   € ð ×%Ñ%ØÔ.¨z¸9ÓEÔGY×GdÑGdó
ð 	
r/   c                óD   — | j                  |dt        j                  «      S )zú
        Adds a tensor to the list of weight tensors to quantize. Called by op quantizers that
        want to quantize a weight (i.e., "mark" a weight for quantization).

        Args:
            tensor_name: name of the weight to quantize
        N)r°   r&   r,   r±   s     r0   Úquantize_weight_tensorz#QDQQuantizer.quantize_weight_tensorR  s    € ð ×%Ñ% k°4Ô9K×9RÑ9RÓSÐSr/   c                ól  — t        || j                  j                  «       «      }|ru|j                  t        j
                  j                  t        j
                  j                  fv r4t        t        j                  ||j                  ¬«      | j                  |<   y y t        j                  d|› d�«       y )N)r<   r>   r@   z9only support per-channel quantization on weight. Tensor: z is not quantized.)r   r�   rš   r@   r¥   r   r¦   r§   r9   r&   r,   rq   r‰   rŠ   )rA   r    r>   r¡   s       r0   Ú"quantize_weight_tensor_per_channelz/QDQQuantizer.quantize_weight_tensor_per_channel\  s“   € Ü˜k¨4¯:©:×+AÑ+AÓ+CÓDˆÙØ×Ñ¤J×$:Ñ$:×$@Ñ$@Ä*×BXÑBX×B`ÑB`Ð#aÑaÜ8JÜ 2× 9Ñ 9ÀÐPV×P`ÑP`ô9�×(Ñ(¨Ò5ð bô
 �O‰OÐWÐXcÐWdÐdvÐwÕxr/   c                ó  — | j                   j                  |j                  «      dz   }|j                  › |› �}t        j                  «       }|j                  |«       ||_        | j                   j                  |«       |S )zk
        Duplicates an existing initializer and adds it to the model. Returns the new initializer.
        r	   )r�   Ú#get_largest_initializer_name_suffixÚnameÚonnxr   ÚCopyFromÚadd_initializer)rA   rš   Úname_suffixÚnew_initializer_nameÚnew_initializers        r0   Ú_dup_initializerzQDQQuantizer._dup_initializerf  sv   € ð  Ÿ:™:×IÑIÈ+×JZÑJZÓ[Ð^_Ñ_ˆØ"-×"2Ñ"2Ð!3°K°=ÐAÐÜ×*Ñ*Ó,ˆØ× Ñ  Ô-Ø3ˆÔØ�
‰
×"Ñ" ?Ô3ØÐr/   c                ó  — | j                   j                  |«      rVt        j                  d|› d�«       | j	                  |d¬«      \  }}|r| j                  ||«       y| j                  |«       yt        || j                  j                  «       «      }|€t        j                  d|› d�«       y|j                  t        j                  j                  t        j                  j                  fvrt        j                  d|› d�«       y|}	|| j                   v rW| j#                  |«      }
|
j$                  }	| j                  j'                  ||	|h«       t        j                  d	|› d
|	› d�«       t)        ||||«      | j                   |	<   y)a¾  
        Adds a bias tensor to the list of bias tensors to quantize. Called by op quantizers that
        want to quantize a bias with bias_zero_point = 0 and bias_scale = input_scale * weight_scale * beta.
        TODO: Explain the reasoning for using this formula.

        Args:
            node_name: name of the node that consumes the bias, input, and weight tensors.
            bias_name: name of the bias tensor to quantize.
            input_name: name of the input tensor whose scale is used to compute the bias's scale.
            weight_name: name of the weight tensor whose scale is used to compute the bias's scale.
            beta: Multiplier used to compute the bias's scale.
        zQuantizing bias tensor 'z=' as a weight due to the presence of user-specified overridesr   )Údefault_axisNzExpected bias 'z' to be an initializerz%' to be an floating-point initializerzCreated a copy of bias input 'z
' called 'ú')Útensor_quant_overridesrt   r‰   ÚinfoÚis_tensor_per_channelr¹   r·   r   r�   rš   rŠ   r@   r¥   r   r¦   r§   rr   rÃ   r¼   Úreplace_input_of_nodesrD   )rA   r5   Ú	bias_namer4   rE   rG   Úis_per_channelr>   Úbias_initializerÚactual_bias_nameÚnew_bias_initializers              r0   Úquantize_bias_tensorz!QDQQuantizer.quantize_bias_tensorr  s…  € ð ×&Ñ&×*Ñ*¨9Ô5Ü�L‰LØ*¨9¨+Ð5rÐsôð $(×#=Ñ#=¸iÐVWÐ#=Ó#XÑ ˆN˜DÙØ×7Ñ7¸	À4ÔHð ð ×+Ñ+¨IÔ6Øä'¨	°4·:±:×3IÑ3IÓ3KÓLÐØÐ#Ü�O‰O˜o¨i¨[Ð8NÐOÔPØà×%Ñ%¬j×.DÑ.D×.JÑ.JÌJ×LbÑLb×LjÑLjÐ-kÑkÜ�L‰L˜?¨9¨+Ð5ZÐ[Ô\Øà$ÐØ˜×-Ñ-Ñ-ð $(×#8Ñ#8Ð9IÓ#JÐ Ø3×8Ñ8Ðð �J‰J×-Ñ-¨iÐ9IÈIÈ;ÔWÜ�L‰LÐ9¸)¸ÀJÐO_ÐN`Ð`aÐbÔcô 3CÀ9ÈjÐZeÐgkÓ2lˆ×ÑÐ.Ò/r/   c                óî  — |j                   syt        |«      }t        j                  t        j                  «      }d}t        j
                  |j                  t        j                  ¬«      t        j
                  |j                  dz   t        j                  ¬«      z
  }	|j                  }
d}|�s”t        j                  |j                  «       t        j
                  dt        j                  ¬«      «      }t        j                  |j                  «       t        j
                  dt        j                  ¬«      «      }t        j                  t        j                  |«      t        j                  |«      «      }|d|z  z  |	z  }t        j
                  |j                  «       t        j                  ¬«      }t        j
                  |j                  «       t        j                  ¬«      }||z  }||k  rK|dkD  rF||z  }t        j                  d	|› d
|› d|j                   › d�«       ||z  }|j#                  |
«      }d}||fS |j$                  �r!t'        |j$                  «      dk(  �r|j$                  d   }t)        |«      D ]ë  }t        j                  ||   «      }|d|z  z  |	z  }t        j
                  |j                  «       t        j                  ¬«      }t        j
                  ||   j                  «       t        j                  ¬«      }||z  }||k  sŒš|dkD  sŒ ||z  }t        j                  d|› d|› d|› d|j                   › d�	«       ||z  }|j#                  |
«      ||<   d}Œí ||fS )aI  
        Checks if the bias scale (input_scale * weight_scale) that we intend to use is too small.
        A bias scale that is too small leads to quantized bias values that fall outside the range of a int32 and have to
        be clipped, which decreases accuracy. If this function detects such a scenario, the weight_scale value will be
        increased to prevent this from happening.

        Although the adjustment method and amount differs, the idea to adjust the weight's scale came from the following
        reference:
        https://github.com/tensorflow/tensorflow/blob/master/tensorflow/lite/tools/optimize/quantization_utils.cc#L252

        :param input_scale: The input's scale.
        :param weight_scale: The weight scale to potentially adjust.
        :param weight_name: The weight initializer's name. Used for logging.
        :param bias_tp: The bias ONNX initializer.
        :param is_per_channel: True if the bias and weight are quantized per-channel.
        :return: A tuple with a bool indicating if the weight's scale was adjusted and the new weight scale.
        ©FNgq¬‹Ûh ð?©Údtyper	   Fr   g       @g        zIncreasing scale for weight `z` by the ratio z to ensure bias input `z` has a valid scale.TzIncreased scale[z] for weight `z` by ratio )Úsizer#   ÚnpÚiinfoÚint32ÚarrayÚmaxÚfloat64ÚminrÔ   ÚminimumÚmaximumÚabsÚitemr‰   rÈ   r¼   ÚastypeÚshapeÚlenÚrange)rA   Úinput_scaleÚweight_scalerE   Úbias_tprÌ   Úbias_float_dataÚ
int32_infoÚmultiplicative_epsilonÚqrangeÚweight_scale_dtypeÚupdated_an_elemÚrminÚrmaxÚabsmaxÚbias_smallest_valid_scaleÚinput_scale_fp64Úweight_scale_fp64Úbias_candidate_scaleÚratioÚ	new_scaleÚ	num_elemsÚiÚ	bias_rmaxs                           r0   Ú#_adjust_weight_scale_for_int32_biasz0QDQQuantizer._adjust_weight_scale_for_int32_bias£  s  € ð2 × Ò Øä/°Ó8ˆä—X‘XœbŸh™hÓ'ˆ
Ø!'ÐÜ—‘˜*Ÿ.™.´·
±
Ô;¼b¿h¹hÀzÇ~Á~ÐXYÑGYÔac×akÑakÔ>lÑlˆØ)×/Ñ/ÐØˆâÜ—:‘:˜o×1Ñ1Ó3´R·X±X¸aÄrÇzÁzÔ5RÓSˆDÜ—:‘:˜o×1Ñ1Ó3´R·X±X¸aÄrÇzÁzÔ5RÓSˆDÜ—Z‘Z¤§¡ t£¬b¯f©f°T«lÓ;ˆFØ(>À#ÈÁ,Ñ(OÐRXÑ(XÐ%ä!Ÿx™x¨×(8Ñ(8Ó(:Ä"Ç*Á*ÔMÐÜ "§¡¨×):Ñ):Ó)<ÄBÇJÁJÔ OÐØ#3Ð6GÑ#GÐ à$Ð'@Ò@ÐG[Ð^aÒGaà1Ð4HÑH�Ü—‘Ø3°K°=ÀÐPUÈwð W*Ø*1¯,©,¨Ð7KðMôð .°Ñ5�	Ø(×/Ñ/Ð0BÓC�Ø"&�ð.  Ð,Ð,ð- ×Ó¤C¨×(:Ñ(:Ó$;¸qÓ$@à$×*Ñ*¨1Ñ-ˆIä˜9Ó%ò +�ÜŸF™F ?°1Ñ#5Ó6�	Ø,BÀcÈIÁoÑ,VÐY_Ñ,_Ð)ä#%§8¡8¨K×,<Ñ,<Ó,>ÄbÇjÁjÔ#QÐ Ü$&§H¡H¨\¸!©_×-AÑ-AÓ-CÌ2Ï:É:Ô$VÐ!Ø'7Ð:KÑ'KÐ$Ø(Ð+DÓDÐK_ÐbeÓKeà5Ð8LÑL�EÜ—L‘LØ*¨1¨#¨^¸K¸=ÈÐTYÐSZð [1Ø18·±°Ð>RðTôð !2°EÑ 9�IØ&/×&6Ñ&6Ð7IÓ&J�L ‘OØ&*‘Oð!+ð$  Ð,Ð,r/   c                ó¶  — | j                   ry| j                  j                  «       D �]®  \  }}|j                  | j                  vs0|j                  | j
                  vs|j                  | j                  vrŒP| j                  |j                     j                  |j                  «      }| j
                  |j                     }t        j                  |d   t        j                  j                  |j                  «      ¬«      }| j                  |j                     }|d   }|t        j                   j"                  t        j                   j$                  fvr�Œ2|d   }|j'                  «       r�ŒI|d   }	|j)                  dd«      du}
| j+                  ||	|j                  t-        || j.                  j1                  «       «      |
«      \  }}|s�Œª||d<   �Œ± y)a3  
        Iterates through all bias inputs that should be quantized to int32. If the intended
        bias scale (equal to input_scale * weight_scale) is too small, this function will increase
        the associated weight's scale to ensure the bias does not overflow the int32 range when quantized.
        NrX   rÓ   Ú
quant_typerY   r>   )r~   rr   Úitemsr4   rŒ   rq   rE   r�   rR   r5   rÖ   Úasarrayr½   ÚhelperÚtensor_dtype_to_np_dtyper@   r   ÚINT8r‚   r…   rt   rú   r   r�   rš   )rA   rË   Ú	bias_infoÚinput_qparamsÚ
input_inforå   Úweight_quant_paramsÚweight_quant_typeÚweight_zero_pointræ   rÌ   Údid_update_weight_scaleÚnew_weight_scales                r0   Ú,_adjust_weight_quant_params_for_bias_tensorsz9QDQQuantizer._adjust_weight_quant_params_for_bias_tensorsó  s¼  € ð ×8Ò8àà$(×$9Ñ$9×$?Ñ$?Ó$Aó &	@Ñ ˆI�yà×$Ñ$¨D×,DÑ,DÑDØ×'Ñ'¨t×/GÑ/GÑGØ×(Ñ(°×0MÑ0MÑMàð !×4Ñ4°Y×5IÑ5IÑJ×[Ñ[Ð\e×\oÑ\oÓpˆMØ×1Ñ1°)×2FÑ2FÑGˆJÜŸ*™*Ø˜gÑ&¬d¯k©k×.RÑ.RÐS]×SgÑSgÓ.hôˆKð #'×"?Ñ"?À	×@UÑ@UÑ"VÐØ 3°LÑ AÐØ ¬×)9Ñ)9×)>Ñ)>Ä×@PÑ@P×@VÑ@VÐ(WÑWÙà,?ÀÑ,MÐØ ×$Ñ$Ô&áà':¸7Ñ'CˆLØ0×4Ñ4°V¸TÓBÈ$ÐNˆNð 9=×8`Ñ8`ØØØ×%Ñ%Ü˜Y¨¯
©
×(>Ñ(>Ó(@ÓAØó9Ñ5Ð#Ð%5ó 'Ø/?Ð# GÓ,ñM&	@r/   c                ó:   — | j                   j                  |«       y r;   )rs   Úappend)rA   Únodes     r0   Úremove_nodezQDQQuantizer.remove_node&  s   € Ø×Ñ×#Ñ# DÕ)r/   c                óN   — | j                   j                  | j                  «       y r;   )r�   Úremove_nodesrs   )rA   s    r0   r  zQDQQuantizer.remove_nodes)  s   € Ø�
‰
×Ñ × 4Ñ 4Õ5r/   c                óÖ  — | j                   j                  «       D ]¯  }| j                  |«      rht        | |«      }|j	                  «        |j
                  D ]=  }|| j                  vrg | j                  |<   | j                  |   j                  |«       Œ? |j                  t        k(  sŒ�|j                  D ]  }|| j                  |<   Œ Œ± | j                  «       | _        | j                  «        | j                  «        | j!                  «        | j"                  r| j%                  «        | j'                  «        | j(                  s| j                   j+                  «        t,        | j                   j                   _        t0        | j                   j                   _        | j4                  t6        k(  r | j                   j9                  t6        d«       | j                   j                   S )Nr	   )r�   ÚnodesÚshould_quantize_noder$   ÚquantizeÚinputry   r  Úop_typer   Úoutputrz   Ú_calc_initializer_quant_paramsr�   r
  Ú_quantize_normal_tensorsÚ_quantize_sharing_param_tensorsrw   Ú_quantize_bias_tensorsr  rv   Úclean_initializersr   Úproducer_namer   Úproducer_versionr|   r   Úset_opset_import)rA   r  Úop_quantizerr    s       r0   Úquantize_modelzQDQQuantizer.quantize_model,  sŒ  € Ø—J‘J×$Ñ$Ó&ò 	DˆDØ×(Ñ(¨Ô.Ü1°$¸Ó=�Ø×%Ñ%Ô'à#'§:¡:ò Q�KØ"¨$×*LÑ*LÑLØJL˜×:Ñ:¸;ÑGØ×6Ñ6°{ÑC×JÑJÈ4ÕPðQð �|‰|œÓ.Ø#'§;¡;ò D�KØ?C�D×/Ñ/°Ò<ñDð	Dð )-×(KÑ(KÓ(MˆÔ%Ø×9Ñ9Ô;Ø×%Ñ%Ô'Ø×,Ñ,Ô.Ø×ÒØ×'Ñ'Ô)Ø×ÑÔØ×*Ò*Ø�J‰J×)Ñ)Ô+ä)5ˆ�
‰
×ÑÔ&Ü,7ˆ�
‰
×ÑÔ)Ø×Ñ¤Ò*Ø�J‰J×'Ñ'¬	°1Ô5à�z‰z×ÑÐr/   c                ó²  — || j                   v rÉ| j                   |   j                  €°| j                   |   j                  €—t        | j                  j	                  «       |   «      dk(  rn| j                  j                  |«      sS| j                  j                  |«      s8| j                  j                  ||«       || j                  v r| j                  |= yy)Nr	   TF)	rŒ   rK   rã   r�   Úinput_name_to_nodesÚis_graph_outputÚis_graph_inputÚreplace_output_of_all_nodesrq   )rA   Úupstream_output_namer´   s      r0   Útry_replacing_upstream_outputz*QDQQuantizer.try_replacing_upstream_outputK  s½   € à˜4×3Ñ3Ñ3Ø×(Ñ(¨Ñ5×?Ñ?ÐGØ×(Ñ(Ð)=Ñ>×HÑHÐPÜ�D—J‘J×2Ñ2Ó4Ð5IÑJÓKÈqÒPØ—J‘J×.Ñ.Ð/CÔDØ—J‘J×-Ñ-Ð.BÔCà�J‰J×2Ñ2Ð3GÈÔUØ# t×'?Ñ'?Ñ?Ø×,Ñ,Ð-AÐBØØr/   c                ó¾   — || j                   dœ}|r||d<   t        j                  j                  t        |||g|g|fi |¤Ž}	| j
                  j                  |	g«       y)zI
        Creates a QuantizeLinear node and adds it to the model.
        ©r>   Údomainr   N)r|   r½   rÿ   Ú	make_noder   r�   Ú	add_nodes)
rA   Úq_inputÚq_outputÚquant_node_nameÚ
scale_nameÚzp_namer>   r   ÚkwargsÚqlinear_nodes
             r0   Ú_create_q_nodezQDQQuantizer._create_q_nodeZ  sj   € ð +/¸$×:LÑ:LÑ!MˆÙØ#-ˆF�<Ñ Ü—{‘{×,Ñ,ÜØ�j 'Ð*ØˆJØñ	
ð
 ñ
ˆð 	�
‰
×Ñ˜l˜^Õ,r/   c                ó¾   — || j                   dœ}|r||d<   t        j                  j                  t        |||g|g|fi |¤Ž}	| j
                  j                  |	g«       y)zK
        Creates a DequantizeLinear node and adds it to the model.
        r*  r   N)r|   r½   rÿ   r,  r   r�   r-  )
rA   Údq_inputÚ	dq_outputÚdequant_node_namer1  r2  r>   r   r3  Údequant_nodes
             r0   Ú_create_dq_nodezQDQQuantizer._create_dq_nodes  sj   € ð +/¸$×:LÑ:LÑ!MˆÙØ#-ˆF�<Ñ Ü—{‘{×,Ñ,ÜØ�z 7Ð+ØˆKØñ	
ð
 ñ
ˆð 	�
‰
×Ñ˜l˜^Õ,r/   c                ó  — |	| j                   dœ}|
r|
|d<   t        j                  j                  t        |||g|g|fi |¤Ž}t        j                  j                  t
        |||g|g|fi |¤Ž}| j                  j                  ||g«       y )Nr*  r   )r|   r½   rÿ   r,  r   r   r�   r-  )rA   r.  r/  r0  r7  r8  r9  r1  r2  r>   r   r3  r4  r:  s                 r0   Ú_create_qdq_nodeszQDQQuantizer._create_qdq_nodesŒ  s¢   € ð +/¸$×:LÑ:LÑ!MˆÙØ#-ˆF�<Ñ Ü—{‘{×,Ñ,ÜØ�j 'Ð*ØˆJØñ	
ð
 ñ
ˆô —{‘{×,Ñ,ÜØ�z 7Ð+ØˆKØñ	
ð
 ñ
ˆð 	�
‰
×Ñ˜l¨LÐ9Õ:r/   c                óØ  — |j                   }|| j                  v ry| j                  |   }|j                  d«      }|j                  dd«      }| j	                  ||«      }d}t        |«      }| j                  j                  ||«       | j                  rat        |«      }	| j                  ||	t        |«      |	|t        |«      |j                  j                   |j                  j                   ||¬«
       n”t        ||d   |d   |d   ||¬«      }
| j                  j!                  |
«       |
j                   }| j#                  |
j                   |t        |«      |j                  j                   |j                  j                   ||¬	«       t%        |||j                  j                   |j                  j                   t&        j(                  |¬
«      }t+        |dd«      | j                  |<   y)a  
        Adds Q/DQ nodes for an initializer. If `self.add_qdq_pair_to_weight` is true, creates
        the sequence (weight_f32 -> Q -> DQ -> ). Otherwise, this function quantizes the initializer
        and adds the sequence (weight_quant -> DQ ->).
        Nr>   r   r   )r   rü   rY   rX   )r>   r   )r>   )r¼   rŽ   r�   rt   Ú_make_scale_zp_initializersr   r�   Úreplace_input_of_all_nodesrv   r   r=  r   r   rX   rY   r!   r¿   r;  r   r   ÚInitializerr]   )rA   Úweight_protorE   Úquant_paramsr>   r   Úscale_zp_initializersÚq_weight_nameÚweight_dequant_outputÚweight_quant_outputÚquant_weightÚquantized_values               r0   Ú_add_qdq_nodes_for_initializerz+QDQQuantizer._add_qdq_nodes_for_initializer¬  sä  € ð #×'Ñ'ˆØ˜$×2Ñ2Ñ2Øà+/×+HÑ+HÈÑ+UˆØ ×$Ñ$ VÓ,ˆØ&×*Ñ*¨<¸Ó;ˆ
Ø $× @Ñ @ÀÈlÓ [ÐØ$(ˆÜ 9¸+Ó FÐØ�
‰
×-Ñ-¨kÐ;PÔQà×&Ò&ô #:¸+Ó"FÐà×"Ñ"ØØ#Ü  Ó-Ø#Ø%Ü" ;Ó/Ø%×+Ñ+×0Ñ0Ø%×0Ñ0×5Ñ5ØØ%ð #õ ô 5ØØ˜\Ñ*Ø˜\Ñ*Ø˜WÑ%ØØ%ôˆLð �J‰J×&Ñ& |Ô4à(×-Ñ-ˆMØ× Ñ Ø×!Ñ!Ø%Ü" ;Ó/Ø%×+Ñ+×0Ñ0Ø%×0Ñ0×5Ñ5ØØ%ð !ô ô )ØØØ!×'Ñ'×,Ñ,Ø!×,Ñ,×1Ñ1Ü×*Ñ*Øô
ˆô 1HÈÐY]Ð_cÓ0dˆ× Ñ  Ò-r/   c                ó  — | j                   �r|| j                  v �r
t        | j                  |   «      dkD  rït        | j                  |   «      }t        |«      D ]È  }d|dz   › �}t	        |«      |z   }t        |«      |z   }	t        |«      |z   }
t        |«      |z   }| j                  |||
||	|||«       | j                  |   |   }| j                  j                  |||	«       |dk(  sŒ�t        ||	||t        j                  |¬«      }t        |d d «      | j                  |<   ŒÊ y |}t        |«      }| j                  j!                  |«      r*t#        |«      }|}| j                  j%                  ||«       n| j                  j'                  ||«       | j                  |t	        |«      t        |«      t	        |«      |t        |«      ||«       t        ||||t        j                  |¬«      }t        |d d «      | j                  |<   y )Nr	   Ú_r   ©Ú
scale_type)rx   ry   rã   rä   r   r   r   r   r=  r�   Úreplace_node_inputr   r   ÚInputr]   rŽ   r$  r   r&  r@  )rA   r    r1  r2  r@   Únum_dedicated_qdq_pairrø   ÚpostfixÚ tensor_name_quant_output_postfixÚ"tensor_name_dequant_output_postfixÚquant_node_name_postfixÚdequant_node_name_postfixr  rI  r.  r8  s                   r0   Ú_add_qdq_pair_for_activationz)QDQQuantizer._add_qdq_pair_for_activationò  s!  € à×#Ó#Ø˜t×AÑAÒAÜ�D×6Ñ6°{ÑCÓDÀqÒHä%(¨×)KÑ)KÈKÑ)XÓ%YÐ"ÜÐ1Ó2ò q�Ø˜a !™e˜W˜+�Ü3JÈ;Ó3WÐZaÑ3aÐ0Ü5NÈ{Ó5[Ð^eÑ5eÐ2Ü*:¸;Ó*GÈ'Ñ*QÐ'Ü,>¸{Ó,KÈgÑ,UÐ)Ø×&Ñ&ØØ4Ø+Ø4Ø6Ø-ØØô	ð ×9Ñ9¸+ÑFÀqÑI�Ø—
‘
×-Ñ-¨d°KÐAcÔdØ˜“6Ü&4Ø#Ø:Ø"ØÜ*×0Ñ0Ø#,ô'�Oô =TÐTcÐeiÐkoÓ<p�D×,Ñ,¨[Ò9ñ9qð< "ˆGÜ1°+Ó>ˆIØ�z‰z×)Ñ)¨+Ô6Ü0°Ó=�Ø'�	Ø—
‘
×6Ñ6°{ÀGÕLà—
‘
×5Ñ5°kÀ9ÔMà×"Ñ"ØÜ'¨Ó4Ü  Ó-Ü'¨Ó4ØÜ" ;Ó/ØØô	ô -ØØØØÜ"×(Ñ(Ø$ôˆOô 5LÈOÐ]aÐcgÓ4hˆD×$Ñ$ [Ò1r/   c                óø  — | j                   j                  |g «      D �ch c]  }|j                  ’Œ }	}| j                  r4|| j                   v r&t	        | j                   |   «      dkD  rt        d«      ‚|	}
|€|	}t        «       }
n|
|z
  }
t	        |«      t	        |	«      k(  }| j                  j                  |«      }|}|r't        |«      }| j                  j                  ||«       t        |«      }| j                  ||t        |«      ||«       t        |«      }|r|s|}|
r"||k7  r| j                  j                  |||
«       | j!                  ||t#        |«      ||«       |}|s/t        |› d�«      }| j!                  ||t#        |› d�«      ||«       t        |› d�«      }| j                  ||t        |› d�«      ||«       t        |› d�«      }|r|r|}|r"||k7  r| j                  j                  |||«       | j!                  ||t#        |› d�«      ||«       t%        ||||t&        j(                  |¬«      }t%        ||||t&        j(                  |¬«      }t+        |||«      | j,                  |<   yc c}w )a†  
        Adds Q and DQ ops to a tensor whose quantized data type is converted. That is, some consumers may use the
        original data type from the producer, while other consumers use the converted data type.
        This is generally done by adding a sequence of ops that convert from one data type (e.g., uint8) to another (e.g., uint16).

        T_float ---> Quant(to u8) ---> Convert(to u16) ---> Dequant(to float) ---> T_float'
        where Convert(to u16) is equivalent to: ---> Dequant(to float) ---> Quant(to u16) --->

        This function handles the following scenarios:

        1) Tensor T is not a graph output; all consumers use the converted type

            <Producer> ---> Q1 ---> DQ1 ---> Q2 ---> DQ2 ---> <Consumers>

        2) Tensor T is not a graph output; some consumers use the original type, others use the converted type

            <Producer> ---> Q1 -+-> DQ1 ---> <Consumers of original type>
                                |
                                +-> DQ1' ---> Q2 ---> DQ2 ---> <Consumers of converted type>

        3) Tensor T is a graph output; all consumers use the converted type

            <Producer> ---> Q1 ---> DQ1 ---> Q2 ---> DQ2 -+-> <Consumers>
                                                          |
                                                          +-> <Graph output>

        4) Tensor T is a graph output; some consumers use the original type, others use the converted type

            <Producer> ---> Q1 -+-> DQ1 -+-> <Consumers of original type>
                                |        |
                                |        +-> <Graph output>
                                |
                                +-> DQ1' ---> Q2 ---> DQ2 ---> <Consumers of converted type>

        5) Tensor T is a graph output that is not consumed by any other nodes.

            <Producer> ---> Q1 ---> DQ1 ---> Q2 ---> DQ2 ---> <Graph output>
        r	   z|Do not currently support converted quant_types in TensorQuantOverrides when the `dedicated_qdq_pair` extra_option is enabledNÚ_convertÚ_convert_clonerM  )ry   rt   r¼   rx   rã   Ú
ValueErrorÚsetr�   r$  r   r&  r   r5  r   r   rÊ   r;  r   r   r   rP  r]   rŽ   )rA   r    Úfirst_scale_nameÚfirst_zp_nameÚscale_data_typeÚconvert_scale_nameÚconvert_zp_nameÚconvert_recv_nodesr  Útensor_recv_nodesÚoriginal_recv_nodesÚall_use_convertedr$  Úfirst_q_inputÚfirst_q_outputÚfirst_dq_outputÚsecond_q_inputÚsecond_q_outputÚsecond_dq_outputÚoriginal_quantized_valueÚconverted_quantized_values                        r0   Ú%_add_qdq_ops_for_converted_activationz2QDQQuantizer._add_qdq_ops_for_converted_activation5  sÕ  € ð` 48×3UÑ3U×3YÑ3YÐZeÐgiÓ3jÖk¨4˜TŸY›YÐkÐÐkð ×#Ò#Ø˜t×AÑAÑAÜ�D×6Ñ6°{ÑCÓDÀqÒHô ð Oóð ð 0ÐØÐ%Ø!2ÐÜ"%£%Ñà"5Ð8JÑ"JÐäÐ 2Ó3´sÐ;LÓ7MÑMÐØŸ*™*×4Ñ4°[ÓAˆð $ˆÙÜ2°;Ó?ˆMØ�J‰J×2Ñ2°;ÀÔNä0°Ó=ˆØ×ÑØ˜>Ô+;¸KÓ+HÐJZÐ\iô	
ô
 4°KÓ@ˆÙÑ#4Ø)ˆOÙ ?°kÒ#AØ�J‰J×-Ñ-¨k¸?ÐL_Ô`à×ÑØ˜OÔ-?ÀÓ-LÐN^Ð`mô	
ð )ˆÙ Ü3°{°mÀ8Ð4LÓMˆNØ× Ñ ØØÜ" k ]°.Ð#AÓBØ Øôô 2°[°MÀÐ2JÓKˆØ×ÑØØÜ ˜}¨HÐ5Ó6ØØô	
ô 5¸°}ÀHÐ5MÓNÐÙÑ0Ø*ÐÙÐ"2°kÒ"AØ�J‰J×-Ñ-¨kÐ;KÐM_Ô`Ø×ÑØØÜ + ¨hÐ7Ó8ØØô	
ô $2ØØØØÜ×$Ñ$Ø&ô$
Ð ô %3ØØØØÜ×$Ñ$Ø&ô%
Ð!ô 1HØ$Ð&?ÐASó1
ˆ× Ñ  Ò-ùòS ls   ŸI7c           
     ó  — | j                   j                  «       j                  «       D �]à  \  }}|| j                  v rŒ|j                  rŒ#t        || j                  j                  «       «      }|r| j                  |«       �nx|| j                  v r| j                   |= Œx| j                  |«      }|st        d|› d�«      ‚|j                  €\| j                  ||j                  j                  j                   |j                  j"                  j                   |j$                  ¬«       nÒ|j$                  |j                  j                  j$                  k(  sJ ‚| j'                  ||j                  j                  j                   |j                  j"                  j                   |j$                  |j                  j                  j                   |j                  j"                  j                   |j(                  «       | j                   |= �Œã y)z}
        Adds Q/DQ ops to tensors (activations and weights) that have been marked for quantization by op quantizers.
        z4Quantization parameters are not specified for param zb. In static mode quantization params for inputs and outputs of nodes to be quantized are required.N)r@   )rq   Úcopyrý   rŽ   r?   r   r�   rš   rJ  rz   Ú"_make_tensor_scale_zp_initializersr[  rK   rW  rJ   rX   r¼   rY   r@   rn  rM   )rA   r    Útensor_inforš   Útensor_qparam_initializerss        r0   r  z%QDQQuantizer._quantize_normal_tensorsÒ  sÕ  € ð )-×(@Ñ(@×(EÑ(EÓ(G×(MÑ(MÓ(Oó 0	:Ñ$ˆK˜Ø˜d×6Ñ6Ñ6Øà×(Ó(ä*¨;¸¿
¹
×8NÑ8NÓ8PÓQ�ÙØ×7Ñ7¸ÖDð # d×&AÑ&AÑAØ ×4Ñ4°[ÐAØ à15×1XÑ1XÐYdÓ1eÐ.Ù5Ü(ØRÐS^ÐR_ð `ð óð ð
 2×;Ñ;ÐCà×9Ñ9Ø'Ø6×?Ñ?×EÑE×JÑJØ6×?Ñ?×JÑJ×OÑOØ&1×&;Ñ&;ð	 :õ ð  +×4Ñ4Ð8R×8[Ñ8[×8aÑ8a×8kÑ8kÒkÐkÐkØ×BÑBØ'Ø6×?Ñ?×EÑE×JÑJØ6×?Ñ?×JÑJ×OÑOØ'×1Ñ1Ø6×@Ñ@×FÑF×KÑKØ6×@Ñ@×KÑK×PÑPØ6×KÑKôð ×,Ñ,¨[Ò9ña0	:r/   c           
     ó°  — | j                   �rÉ| j                   j                  «       j                  «       D �]Ž  \  }}|j                  }|sŒ|j                  | j
                  v sŒ/| j                   |= | j
                  |j                     j                  |j                  «      }| j                  |«      rt        d«      ‚|| j                  v rt        d|› d�«      ‚d}d}|| j                  v rD| j                  |   }|j                  r)| j                  ||j                  d«      }|j                  }|€)| j                  ||j                   |j"                  «       �Œ(| j%                  ||j                   |j"                  |j&                  j(                  |j&                  j*                  |j,                  j*                  |«       �Œ‘ | j                   r�ŒÈyy)a{  
        Adds Q/DQ ops to tensors that have been marked for quantization by op quantizers.
        Only operates on tensors that want to use the quantization parameter initializers from an upstream tensor.
        For example, a Transpose node's output tensor will typically want to use the same quantization parameter
        initializers as the Transpose node's input.
        zBQuantization parameter shared mode is not supported for weight yetz5Quantization parameter sharing is invalid for tensor z& because it has already been quantizedNrY  )rq   rp  rý   r=   r4   rŽ   rR   r5   Úis_input_a_initializerr[  rz   rŒ   rK   r?  rM   rW  r1  r2  rn  rX   r@   r¼   rY   )rA   r    rr  Úquant_providerrI  Úconverted_qparam_initsrM   Útensor_paramss           r0   r  z,QDQQuantizer._quantize_sharing_param_tensors  sÉ  € ð ×&Ó&Ø,0×,DÑ,D×,IÑ,IÓ,K×,QÑ,QÓ,Só .Ñ(�˜[Ø!,×!@Ñ!@�Ú! n×&?Ñ&?À4×C[ÑC[Ò&[Ø×0Ñ0°Ð=à&*×&>Ñ&>¸~×?XÑ?XÑ&Y×&jÑ&jØ&×0Ñ0ó'�Oð ×2Ñ2°;Ô?Ü(Ð)mÓnÐnà" d×&AÑ&AÑAÜ(ØSÐT_ÐS`ð aDð Dóð ð .2Ð*Ø+/Ð(Ø" d×&>Ñ&>Ñ>Ø(,×(@Ñ(@ÀÑ(M˜Ø(×2Ò2Ø59×5UÑ5UØ +¨]×-DÑ-DÀjó6Ð2ð 4A×3UÑ3UÐ0à-Ð5à×9Ñ9Ø'¨×)CÑ)CÀ_×E\ÑE\öð ×BÑBØ'Ø+×6Ñ6Ø+×3Ñ3Ø2×8Ñ8×BÑBØ2×8Ñ8×=Ñ=Ø2×=Ñ=×BÑBØ0öðM.ð ×&Ö&r/   c           	     ó.  — | j                   j                  «       D �]w  \  }}|| j                  v rŒ| j                  ||«       t	        || j
                  j                  «       «      }| j
                  j                  |«       | j                  |   j                  }|j                  dk(  r�t        |j                  t        «      s.t        dt        |j                  «      › d|j                  ›�«      ‚t!        |«      }t"        j$                  j'                  d|j(                  g|g||j                  ¬«      }�n?|j                  dv �r|j*                  t"        j,                  j.                  t"        j,                  j0                  t"        j,                  j2                  hv rt5        d|j*                  › d�«      ‚|j(                  |j6                  |j8                  g}t!        |«      }|j:                  �;t"        j$                  j'                  d	||g||j:                  | j<                  ¬
«      }nIt"        j$                  j'                  d	||g|| j<                  ¬«      }nt5        d|j                  ›d�«      ‚| j
                  j?                  |«       �Œz y)zq
        Adds DQ ops (or Cast) for bias tensors that have been marked for quantization by op quantizers.
        ÚCastúUnexpected type z for input=)r¼   Úto)NÚDequantizeLinearzUnexpected quantize type z for DequantizeLinear.Nr}  r*  )r+  zUnexpected operator type rª   ) rr   rý   rŽ   Úquantize_bias_staticr   r�   rš   Úremove_initializerrJ   Ú	node_typer«   r@   Úintr¬   rœ   r4   r   r½   rÿ   r,  Úq_nameÚ
node_qtyper   r§   ÚBFLOAT16r¦   ÚRuntimeErrorr1  r2  r>   r|   Úadd_node)rA   rË   r  ÚinitÚquant_valuer5   r:  Úinputss           r0   r  z#QDQQuantizer._quantize_bias_tensors@  sT  € ð %)×$9Ñ$9×$?Ñ$?Ó$Aó 1	.Ñ ˆI�yØ˜D×4Ñ4Ñ4Øà×%Ñ% i°Ô;Ü 	¨4¯:©:×+AÑ+AÓ+CÓDˆDØ�J‰J×)Ñ)¨$Ô/Ø×2Ñ2°9Ñ=×FÑFˆKØ×$Ñ$¨Ò.ô " $§.¡.´#Ô6Ü#Ð&6´t¸D¿N¹NÓ7KÐ6LÈKÐXa×XlÑXlÐWoÐ$pÓqÐqÜ.¨yÓ9�	Ü#Ÿ{™{×4Ñ4ØØ ×'Ñ'Ð(Ø�KØ"Ø—~‘~ð  5ó  ’ð ×&Ñ&Ð*DÒDØ×)Ñ)Ü×$Ñ$×,Ñ,Ü×$Ñ$×-Ñ-Ü×$Ñ$×*Ñ*ð.ñ ô
 'Ð)BÀ;×CYÑCYÐBZÐZpÐ'qÓrÐrØ%×,Ñ,¨k×.DÑ.DÀk×FYÑFYÐZ�Ü.¨yÓ9�	Ø×#Ñ#Ð/Ü#'§;¡;×#8Ñ#8Ø*ØØ"˜Ø!Ø(×-Ñ-Ø#×1Ñ1ð $9ó $‘Lô $(§;¡;×#8Ñ#8Ø*ØØ"˜Ø!Ø#×1Ñ1ð $9ó $‘Lô #Ð%>¸{×?TÑ?TÐ>WÐWXÐ#YÓZÐZØ�J‰J×Ñ Ö-ñc1	.r/   c                ó>   — || j                   v xs || j                  v S r;   )rq   rr   r±   s     r0   Úis_tensor_quantizedz QDQQuantizer.is_tensor_quantizedw  s#   € Ø˜d×6Ñ6Ð6Ò^¸+È×I^ÑI^Ð:^Ð^r/   c                óæ  — | j                   j                  |«      }|€y| j                  j                  |«      ry| j                  j	                  |«      }| j
                  s|sy|r| j                  j                  ||«      n|}|r#| j                  j                  |«      }|d   d   }t        |j                  «      }t        ||«      \  }	}|	st        j                  d|› d|› d|› �«       yd|fS )aÞ  
        Checks if a given tensor is configured to be quantized per-channel. If so, also returns the channel axis.

        ORT only supports per-channel quantization on static weights (i.e., ONNX initializers). If the user did not provide
        tensor quantization overrides for this tensor, then the value of self.per_channel determines if the weight
        is to be quantized per-channel.

        Params:
            tensor_name: The name of the tensor to check.
            default_axis: The default channel axis. This method checks if the normalized axis is within bounds.
                          Can be overridden via the extra_options 'QDQOpTypePerChannelSupportToAxis'
                          and 'TensorQuantOverrides'.
            op_type: Optional, defaults to None. The operator type that is the only consumer of this weight.
                     Used to access the extra option 'QDQOpTypePerChannelSupportToAxis'.
        Returns:
            A tuple (is_per_channel, axis) in which the first element indicates whether the tensor is
            quantized per-channel and the second element is the channel axis.
            The returned axis is only None if the tensor is not per-channel or the axis is out of bounds.
        rÒ   r   r>   zAxis z is out-of-range for weight 'z' with rank T)Úinitializersrt   rÇ   Úhas_per_tensor_overridesÚhas_per_channel_overridesr�   r{   Úget_per_channel_overridesrã   Údimsr    r‰   rŠ   )
rA   r    rÅ   r  Úweight_initializerÚhas_per_chan_overridesr>   Úper_chan_overridesÚweight_rankÚ
axis_valids
             r0   rÉ   z"QDQQuantizer.is_tensor_per_channelz  sþ   € ð2 "×.Ñ.×2Ñ2°;Ó?ÐØÐ%Øà×&Ñ&×?Ñ?ÀÔLØà!%×!<Ñ!<×!VÑ!VÐWbÓ!cÐØ×ÒÑ(>ØáZaˆt×;Ñ;×?Ñ?ÀÈÔVÐgsˆÙ!Ø!%×!<Ñ!<×!VÑ!VÐWbÓ!cÐØ% aÑ(¨Ñ0ˆDäÐ,×1Ñ1Ó2ˆÜ)¨$°Ó<Ñˆ
�DÙÜ�O‰O˜e D 6Ð)FÀ{ÀmÐS_Ð`kÐ_lÐmÔnØà�TˆzÐr/   c                óL  — | j                   j                  «       }d}|| j                  v r5| j                  |   j                  |«      j                  }t        ||«      }n7| j                  j                  |d«      }|rt        |j                  d   |«      }|�t        |«      S dS )aø  
        Returns the quantization scale of a tensor that is consumed by the given node.
        :parameter tensor_name: The name of the tensor.
        :parameter consumer_node_name: The name of the node that consumes the tensor as input. Necessary in case
                                       the quantization type of the tensor was converted.
                                       Refer: QDQQuantizer::_add_qdq_ops_for_converted_activation.
        :returns: The quantization scale or None.
        Nr	   )
r�   rš   rŽ   rR   r1  r   rz   rt   r  r#   )rA   r    rQ   r�  Úscale_initializerr1  Údq_nodes          r0   Ú_get_tensor_quantization_scalez+QDQQuantizer._get_tensor_quantization_scale«  s¤   € ð —z‘z×-Ñ-Ó/ˆØ59Ðà˜$×2Ñ2Ñ2à×1Ñ1°+Ñ>×OÑOÐPbÓc×nÑnˆJÜ ,¨Z¸Ó FÑð ×1Ñ1×5Ñ5°kÀ4ÓHˆGÙÜ$0°·±¸qÑ1AÀ<Ó$PÐ!à;LÐ;XÔ$Ð%6Ó7ÐbÐ^bÐbr/   c           
     óZ  — || j                   v r#| j                   |   j                  j                  S | j                  |j                  |j
                  «      }|€t        d|j                  › d|› d�«      ‚| j                  |j                  |j
                  «      }|€t        d|j                  › d|› d�«      ‚| j                  ||||j                  «      \  }}}}}	}
t        ||||t        j                  |j                  dkD  rdnd|	|
¬«      }t        |dd«      | j                   |<   |S )	z]
        Quantized the bias. Zero Point == 0 and Scale == Input_Scale * Weight_Scale
        Nz9Unable to get valid quantization scale for weight input 'z' when quantizing bias 'z' to int32.z2Unable to get valid quantization scale for input 'r	   r   )r€  rƒ  )rŽ   rJ   r‚  rš  rE   r5   r[  r4   Úquantize_bias_static_implrG   r   r   rA  rÕ   r]   )rA   rË   r  ræ   rå   Úquantized_bias_nameÚquantized_bias_scale_nameÚquantized_bias_zp_nameÚbias_scale_datar€  rƒ  rI  s               r0   r~  z!QDQQuantizer.quantize_bias_staticÃ  sj  € ð ˜×0Ñ0Ñ0Ø×+Ñ+¨IÑ6×?Ñ?×FÑFÐFð ×:Ñ:¸9×;PÑ;PÐR[×ReÑReÓfˆØÐÜØKÈI×LaÑLaÐKbð c)Ø)2¨°;ð@óð ð ×9Ñ9¸)×:NÑ:NÐPY×PcÑPcÓdˆØÐÜØDÀY×EYÑEYÐDZð [)Ø)2¨°;ð@óð ð ×*Ñ*¨9°kÀ<ÐQZ×Q_ÑQ_Ó`ñ	
ØØ%Ø"ØØØô )ØØØ%Ø"Ü×*Ñ*Ø ×%Ñ%¨Ò)‰A¨tØØ!ô	
ˆô /FÀoÐW[Ð]aÓ.bˆ× Ñ  Ñ+à"Ð"r/   c                óp  — |d   }|d   }|d   }|j                  d«      }|j                  dd«      }t        |«      }	|	r|�t        |j                  «      dk(  s?|	s|�t        |j                  «      dk(  s#|	s|€t        |j                  «      dk(  sJ d	«       ‚t        |j                  «      t        |j                  «      k(  sJ d
«       ‚|dz   |z   }
|dz   |z   }t        j
                  j                  |
||j                  |j                  «       j                  «       «      }| j                  j                  |«       |j                  t        j                  k(  rt        j                  j                   }nS|j                  t        j"                  k(  rt        j                  j$                  }nt'        d|j                  › d|›�«      ‚t        j
                  j                  |||j                  |j                  «       j                  «       «      }| j                  j                  |«       t)        ||«      S )zù
        Creates and returns scale and zero-point initializers for the given quantization params. The initializers are
        named:
            - {param_name}_zero_point{init_name_suffix}
            - {param_name}_scale{init_name_suffix}
        rY   rX   rü   r>   r   r   r'   r	   zWrong scale/zp shapesz,Scale and zero-point must have the same rankÚ_zero_pointÚ_scalezUnexpected dtype=z for param_name=)rt   Úboolrã   râ   r½   rÿ   Úmake_tensorÚravelÚtolistr�   r¿   rÔ   rÖ   Úfloat32r¥   r   r¦   Úfloat16r§   r[  rW   )rA   Ú
param_namerC  Úinit_name_suffixrY   rX   Úzero_point_typer>   r   Ú
is_blockedÚzero_point_namer1  Úinit_zprN  Ú
init_scales                  r0   r?  z(QDQQuantizer._make_scale_zp_initializersó  sï  € ð " ,Ñ/ˆ
Ø˜WÑ%ˆØ& |Ñ4ˆØ'×+Ñ+¨FÓ3ˆØ&×*Ñ*¨<¸Ó;ˆ
Ü˜*Ó%ˆ
á˜DÐ,´°U·[±[Ó1AÀQÒ1FÙ 4Ð#3¼¸E¿K¹KÓ8HÈAÒ8MÙ 4 <´C¸¿¹Ó4DÈÒ4Ið	#ð #ó		#ðKô �5—;‘;Ó¤3 z×'7Ñ'7Ó#8Ò8ÐhÐ:hÓhÐ8à$ }Ñ4Ð7GÑGˆØ (Ñ*Ð-=Ñ=ˆ
ô —+‘+×)Ñ)Ø˜_¨j×.>Ñ.>À
×@PÑ@PÓ@R×@YÑ@YÓ@[ó
ˆð 	�
‰
×"Ñ" 7Ô+à�;‰;œ"Ÿ*™*Ò$Ü#×/Ñ/×5Ñ5‰JØ�[‰[œBŸJ™JÒ&Ü#×/Ñ/×7Ñ7‰JäÐ0°·±°Ð=MÈjÈ^Ð\Ó]Ð]Ü—[‘[×,Ñ,¨Z¸ÀUÇ[Á[ÐRW×R]ÑR]ÓR_×RfÑRfÓRhÓiˆ
Ø�
‰
×"Ñ" :Ô.ä% j°'Ó:Ð:r/   c                óš  — | j                   �|| j                   vrt        j                  d|› d�«       y| j                   |   }t        |t        «      st        dt        |«      › d|›d�«      ‚| j                  ||j                  «      }|j                  r| j                  ||j                  d«      nd}t        |||j                  «      S )a  
        Create and returns all scale/zero_point initializers for a given tensor. If the tensor is converted
        to a different quantization type, this function creates two pairs of zp/scale initializers. Otherwise,
        only one pair of zp/scale initializers is created.
        Nz$Quantization parameters for tensor:"z" not specifiedr{  ú for rª   rY  )rŒ   r‰   rÈ   r«   rI   r¬   rœ   r?  rJ   rK   r[   rM   )rA   r    rx  Úoriginal_initsÚconverted_initss        r0   rq  z/QDQQuantizer._make_tensor_scale_zp_initializers  sÐ   € ð ×#Ñ#Ð+¨{À$×BZÑBZÑ/ZÜ�L‰LÐ?À¸}ÈOÐ\Ô]Øà×0Ñ0°Ñ=ˆÜ˜-Ô)=Ô>ÜÐ.¬t°MÓ/BÐ.CÀ5ÈÈÐWXÐYÓZÐZà×9Ñ9¸+À}×G]ÑG]Ó^ˆð ×&Ò&ð ×,Ñ,¨[¸-×:QÑ:QÐS]Ô^àð 	ô ,¨N¸OÈ]×MoÑMoÓpÐpr/   c                óô  — | j                   }d|v r|d   j                  }d|v rd|v r|d   |d   }}�n|t        j                  j                  k(  rt        ||j                  d   «      \  }}nâ|j                  d|j                  d   «      }|j                  d|j                  d   «      }|j                  d| j                  «      }|j                  d	d
«      }	t        ||	|¬«      \  }
}t        |||
||| j                  «      \  }}| j                  r<|t        j                  j                  k(  r|st        |||
|| j                  ¬«      \  }}t!        |j#                  «       |j#                  «       |¬«      S )z”
        Calculates quantization parameters (scale/zero-point) given a tensor's min/max range and optional
        user-provided overrides.
        rü   rX   rY   r	   rî   r   rï   Ú	symmetricr‘   F)r‘   r¶  )ÚqminÚqmaxÚmin_real_range©rY   rX   rü   )r‡   r<   r½   r   ÚFLOAT8E4M3FNr   Úavg_stdrt   Úrange_valueÚis_activation_symmetricr   r   r¹  Ú#is_activation_restricted_asymmetricÚUINT8r"   r   Úsqueeze)rA   Útensor_dataÚquant_overridesrü   ÚzerorX   rî   rï   r¶  r‘   r·  r¸  s               r0   Úcalc_quant_paramszQDQQuantizer.calc_quant_params4  sj  € ð
 ×*Ñ*ˆ
Ø˜?Ñ*Ø(¨Ñ6×BÑBˆJà�oÑ%¨,¸/Ñ*IØ)¨,Ñ7¸ÈÑ9Q�%ŠDØœ4×+Ñ+×8Ñ8Ò8Ü1°*¸k×>QÑ>QÐRSÑ>TÓU‰KˆD‘%à"×&Ñ& v¨{×/FÑ/FÀqÑ/IÓJˆDØ"×&Ñ& v¨{×/FÑ/FÀqÑ/IÓJˆDØ'×+Ñ+¨K¸×9UÑ9UÓVˆIØ*×.Ñ.¨~¸uÓEˆLÜ0°È,ÐbkÔl‰JˆD�$Ü*¨4°°t¸TÀ9Èd×NaÑNaÓb‰KˆD�%Ø×7Ò7¸JÌ$×JZÑJZ×J`ÑJ`Ò<`Ñirä6Ø˜$ T°ÀT×EXÑEXô‘��eô "¨T¯\©\«^À5Ç=Á=Ã?Ð_iÔjÐjr/   c                ó¼  — | j                   €i S | j                  «        i }| j                   D ]¬  }| j                   |   }t        |t        «      st	        dt        |«      › d|›d�«      ‚| j                  j                  |i ¬«      }| j                  ||«      }d}d}d|v r)| j                  ||d   «      }|d   j                  d«      }t        |||«      ||<   Œ® |S )z´
        Calculates quantization parameters (scale/zero-point) for all tensors in the graph using each tensor's min/max range
        and optional user-provided overrides.
        Nr{  r²  rª   )Údefault_valÚconvertÚ
recv_nodes)r’   Úadjust_tensor_rangesr«   r   r¬   rœ   rÇ   Úget_per_tensor_overridesrÅ  rt   rI   )rA   rŒ   r    ÚtdrÃ  rJ   rK   rM   s           r0   r‹   z$QDQQuantizer.calc_graph_quant_paramsP  s  € ð
 ×ÑÐ%ØˆIà×!Ñ!Ô#à ÐØ×-Ñ-ò 	oˆKØ×#Ñ# KÑ0ˆBÜ˜b¤*Ô-ÜÐ"2´4¸³8°*¸EÀ+ÀÐPQÐ RÓSÐSà"×9Ñ9×RÑRÐS^ÐlnÐRÓoˆOØ×-Ñ-¨b°/ÓBˆHØˆIØ#'Ð à˜OÑ+Ø ×2Ñ2°2°ÀyÑ7QÓR�	Ø'6°yÑ'A×'EÑ'EÀlÓ'SÐ$ä/CÀHÈiÐYmÓ/nÐ Ò,ð	oð  #Ð"r/   c                óÐ	  — i }| j                   j                  «       D �]Å  \  }}t        || j                  j	                  «       «      }|sŒ.t        |«      }t        |j                  «      }|j                  t        j                  u }|r| j                  n| j                  }| j                  j                  |«      �rb| j                  |   }	d|	d   v r|	d   d   j                  }t        |   }
d|	d   v }|sQt!        t#        j$                  |	d   d   |
¬«      t#        j$                  |	d   d   |j&                  «      |¬«      ||<   nÕg }g }|	D ]]  }|j)                  t#        j$                  |d   |
«      «       |j)                  t#        j$                  |d   |j&                  ¬«      «       Œ_ |	d   d   }t+        ||«      \  }}|st-        d|j.                  › d	|› d
|› �«      ‚t!        t#        j$                  |«      t#        j$                  |«      ||¬«      ||<   �Œ| j                  j1                  |i g«      }	d|	d   v r|	d   d   j                  }|	d   j1                  d|j2                  «      }|du}|xs |r| j5                  |«      n| j6                  }|	d   j1                  d|«      }|	d   j1                  d| j8                  «      }d}d}|�|nd}|r{| j:                  dkD  rlt+        ||«      \  }}|st-        d|j.                  › d|› d
|› �«      ‚|}t=        |||| j:                  |«      \  }}t!        ||||| j:                  ¬«      ||<   �Œ9|set?        |jA                  «       |||| jB                  |	d   j1                  d«      |	d   j1                  d«      ¬«      \  }}t!        ||||¬«      ||<   �Œ t+        ||«      \  }}|st-        d|j.                  › d	|› d
|› �«      ‚|}|j                  |   }g }g }tE        |«      D ]˜  }|jG                  ||«      }|	r|t        |	«      k  r|	|   ni }t?        |jI                  «       |||| jB                  |j1                  d«      |j1                  d«      ¬«      \  }}|j)                  |«       |j)                  |«       Œš t#        jJ                  |«      }t#        jJ                  |«      }t!        ||||¬«      ||<   �ŒÈ |S )ze
        Returns quantization parameters (scale/zero_point/quant_type) for all initializers.
        rü   r   r>   rY   rÓ   rX   rº  zWeight z# has a per-channel axis with value z  that is out-of-bounds for rank )rY   rX   rü   r>   Nr¶  r‘   z" has a block-wise axis with value )rY   rX   rü   r>   r   rî   rï   )r‘   r¹  Úrmin_overrideÚrmax_override)&rq   rý   r   r�   rš   r#   rã   râ   r<   r&   r,   rˆ   r‡   rÇ   Úoverrides_scale_zpr   r   rÖ   rÙ   rÔ   r  r    r[  r¼   rt   r>   Úis_weight_symmetricr¾  r‘   r   r   r   Úflattenr¹  rä   Útaker¦  rþ   )rA   rŒ   r    rr  rš   Úinitializer_dataÚinitializer_rankÚ	is_weightrü   Ú	overridesÚzp_dtyperÌ   Úzero_points_listÚscales_listÚchan_overridesÚchannel_axisÚis_axis_validÚnorm_channel_axisÚis_symmetric_defaultÚis_symmetricr‘   rY   rX   Ú
block_axisÚnorm_block_axisÚchannel_countrø   Úper_channel_dataÚchannel_overridesÚchannel_zero_pointÚchannel_scales                                  r0   r  z+QDQQuantizer._calc_initializer_quant_paramsm  s‹  € ð
 >@ÐØ(,×(@Ñ(@×(FÑ(FÓ(Hó P	Ñ$ˆK˜Ü& {°D·J±J×4JÑ4JÓ4LÓMˆKÙØä4°[ÓAÐÜ"Ð#3×#9Ñ#9Ó:Ðð $×/Ñ/Ô3E×3LÑ3LÐLˆIÙ.7˜×*Ò*¸T×=RÑ=RˆJð ×*Ñ*×=Ñ=¸kÕJØ ×7Ñ7¸ÑD�	Ø 9¨Q¡<Ñ/Ø!*¨1¡¨lÑ!;×!GÑ!G�Jä/°
Ñ;�Ø!'¨9°Q©<Ð!7�Ù%Ü7IÜ#%§8¡8¨I°a©L¸Ñ,FÈhÔ#WÜ Ÿh™h y°¡|°GÑ'<Ð>N×>TÑ>TÓUØ#-ô8Ð'¨Ò4ð (*Ð$Ø"$�KØ*3ò l˜Ø(×/Ñ/´·±¸ÈÑ9UÐW_Ó0`ÔaØ#×*Ñ*¬2¯8©8°NÀ7Ñ4KÐSc×SiÑSiÔ+jÕkðlð $-¨Q¡<°Ñ#7�LÜ7EÀlÐTdÓ7eÑ4�MÐ#4Ù(Ü(Ø% k×&6Ñ&6Ð%7Ð7ZÐ[gÐZhð i6Ø6FÐ5GðIóð ô
 8JÜ#%§8¡8Ð,<Ó#=Ü Ÿh™h {Ó3Ø#-Ø.ô	8Ð'¨Ñ4ñ ð ×3Ñ3×7Ñ7¸ÀbÀTÓJˆIØ˜y¨™|Ñ+Ø& q™\¨,Ñ7×CÑC�
à$ Q™<×+Ñ+¨F°K×4DÑ4DÓEˆLØ)°Ð5ˆNð $2ò $Ù8A�×(Ñ(¨Ô4Àt×GcÑGcð !ð % Q™<×+Ñ+¨KÐ9MÓNˆLØ$ Q™<×+Ñ+¨N¸D×<MÑ<MÓNˆLØ,0ˆJØ'+ˆEð *6Ð)A™ÀqˆJÙ˜TŸ_™_¨qÒ0ä1?À
ÐL\Ó1]Ñ.�˜Ù$Ü$Ø! +×"2Ñ"2Ð!3Ð3UÐV`ÐUað b2Ø2BÐ1CðEóð ð  /�Ü$<Ø$ØØ Ø—O‘OØ ó%Ñ!�
˜Eô 4FØ)ØØ)Ø%Ø#Ÿ™ô4Ð# KÓ0ñ $Ü$=Ø$×,Ñ,Ó.ØØ Ø!-Ø#'×#6Ñ#6Ø"+¨A¡,×"2Ñ"2°6Ó":Ø"+¨A¡,×"2Ñ"2°6Ó":ô%Ñ!�
˜Eô 4FØ)ØØ)Ø%ô	4Ð# KÓ0ô 4BÀ,ÐP`Ó3aÑ0�Ð0Ù$Ü$Ø! +×"2Ñ"2Ð!3Ð3VÐWcÐVdð e2Ø2BÐ1CðEóð ð
  1�Ø 0× 6Ñ 6°|Ñ D�Ø#%Ð Ø �Ü˜}Ó-ò 6�AØ'7×'<Ñ'<¸QÀÓ'MÐ$Ù8AÀaÌ#ÈiË.ÒFX¨	°!ªÐ^`Ð%Ü8QØ(×.Ñ.Ó0Ø"Ø$Ø%1Ø'+×':Ñ':Ø&7×&;Ñ&;¸FÓ&CØ&7×&;Ñ&;¸FÓ&Cô9Ñ5Ð&¨ð %×+Ñ+Ð,>Ô?Ø×&Ñ& }Õ5ð6ô  ŸZ™ZÐ(8Ó9�
ÜŸ
™
 ;Ó/�Ü3EØ)ØØ)Ø%ô	4Ð# KÓ0ðWP	ðd #Ð"r/   r;   )r    r3   )r´   r3   r4   r3   r5   r3   )rš   úonnx.TensorProtorT   rè  )g      ð?)rå   ú
np.ndarrayræ   ré  rE   r3   rç   rè  rÌ   r¤  rT   ztuple[bool, np.ndarray | None])Nr   )r.  r3   r/  r3   r0  r3   r1  r3   r2  r3   r>   ú
int | Noner   r�  )r7  r3   r8  r3   r9  r3   r1  r3   r2  r3   r>   rê  r   r�  )r   r�  )rB  rè  )r    r3   rÅ   r�  r  z
str | NonerT   ztuple[bool, int | None])r    r3   rQ   r3   rT   znp.ndarray | None)rË   r3   r  rD   rT   r3   )Ú )rª  r3   rC  r   r«  r3   rT   rW   )r    r3   rT   z#QDQTensorScaleZpInitializers | None)rÂ  r   rÃ  zdict[str, Any]rT   r   )rT   zdict[str, QDQTensorQuantParams])rT   zdict[str, QuantizationParams])'r(   r)   r*   rB   r£   r¨   r&   r+   r°   r²   rµ   r·   r¹   rÃ   rÐ   rú   r
  r  r  r!  r(  r5  r;  r=  rJ  rW  rn  r  r  r  r‹  rÉ   rš  r~  r?  rq  rÅ  r‹   r  r.   r/   r0   r`   r`   ‘   s  „ ð óc&òJòð, EIÐVh×VsÑVsó yó:Xó
ó"Tòyó
ó/mðbN-àðN-ð !ðN-ð ð	N-ð
 "ðN-ð ðN-ð 
(óN-ò`1@òf*ò6ò ò>ð,  Øð-àð-ð ð-ð ð	-ð
 ð-ð ð-ð ð-ð ó-ð@  Øð-àð-ð ð-ð ð	-ð
 ð-ð ð-ð ð-ð ó-ðF Øð;ð ó;ó@DeóLAiòF[
òz4:òl6òp5.ón_ð #ð	/àð/ð ð/ð ð	/ð
 
!ó/óbcó0.#ðb Z\ð(;Øð(;Ø-?ð(;ØSVð(;à	ó(;óTqó.kó8#ô:X#r/   r`   )7Ú
__future__r   r‰   Údataclassesr   Úenumr   Útypingr   ÚnumpyrÖ   r½   r   r   r¥   Úbase_quantizerr
   r   Ú	calibrater   Úquant_utilsr   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r    r!   r"   r#   Úregistryr$   r&   r2   r9   rD   rI   rW   r[   r]   r`   r.   r/   r0   ú<module>rõ     s  ðõ #ã Ý !Ý Ý ã Û Ý Ý &ç =Ý !÷÷ ÷ ÷ ÷ ÷ ñ õ2 )ô˜ô ð ÷ð ó ð÷#ñ #ð ÷ð ó ðð ÷fð fó ðfð& ÷ð ó ðð ÷*ð *ó ð*ð ÷fð fó ðfô$t#�=õ t#r/   