§
    kŠtj( ã                  ó@  — d dl mZ d dlZd dlmZ d dlmZ d dlmZ d dl	Z
d dlZd dlmZ d dlmZ dd	lmZmZ dd
lmZ ddlmZmZmZmZmZmZmZmZmZmZmZm Z m!Z!m"Z"m#Z#m$Z$m%Z%m&Z&m'Z'm(Z(m)Z)m*Z*m+Z+ ddl,m-Z-  G d„ de¦  «        Z.e G d„ d¦  «        ¦   «         Z/ G d„ d¦  «        Z0e G d„ d¦  «        ¦   «         Z1e G d„ d¦  «        ¦   «         Z2e G d„ d¦  «        ¦   «         Z3e G d„ d¦  «        ¦   «         Z4e G d„ d¦  «        ¦   «         Z5 G d„ de¦  «        Z6dS )é    )ÚannotationsN)Ú	dataclass)ÚEnum)ÚAny)ÚTensorProto)Úonnx_pbé   )ÚBaseQuantizerÚQuantizationParams)Ú
TensorData)ÚDEQUANT_OP_NAMEÚONNX_TYPE_TO_NP_TYPEÚQUANT_OP_NAMEÚQuantizedValueÚQuantizedValueTypeÚ__producer__Ú__version__Úadd_dequant_output_suffixÚadd_dequant_suffixÚadd_quant_input_suffixÚadd_quant_output_suffixÚadd_quant_suffixÚcompute_data_quant_paramsÚcompute_scale_zpÚcompute_scale_zp_blockedÚcompute_scale_zp_float8Úfind_by_nameÚget_qmin_qmax_for_qTypeÚ	ms_domainÚnormalize_axisÚquantize_onnx_initializerÚsnap_zero_point_to_uint8Útensor_proto_to_array)ÚCreateQDQQuantizerc                  ó   — e Zd ZdZdZdZdS )ÚQDQQuantTensorTyper   r	   é   N)Ú__name__Ú
__module__Ú__qualname__Ú
ACTIVATIONÚWEIGHTÚBIAS© ó    úd/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/onnxruntime/quantization/qdq_quantizer.pyr&   r&   0   s   € € € € € Ø€JØ€FØ€D€D€Dr/   r&   c                  ó$   — e Zd ZU ded<   ded<   dS )ÚQDQQuantParamProviderÚstrÚ
input_nameÚ	node_nameN©r(   r)   r*   Ú__annotations__r.   r/   r0   r2   r2   9   s"   € € € € € € à€O€O�OØ€N€N�N€N€Nr/   r2   c                  ó(   — e Zd Zej        dddfd„ZdS )ÚQDQTensorQuantInfoNc                óX   — || _         || _        || _        |d u| _        |€J ‚|| _        d S ©N)Útensor_typeÚquant_para_providerÚaxisÚ	is_sharedÚ	data_type)Úselfr<   r=   r>   r@   s        r0   Ú__init__zQDQTensorQuantInfo.__init__B   s<   € Ø&ˆÔØ#6ˆÔ ØˆŒ	Ø,°DÐ8ˆŒØÐ$Ð$Ð$Ø"ˆŒˆˆr/   )r(   r)   r*   r&   r+   rB   r.   r/   r0   r9   r9   A   s7   € € € € € Ø#5Ô#@ÐVZÐaeÐquð #ð #ð #ð #ð #ð #r/   r9   c                  ó8   — e Zd ZU ded<   ded<   ded<   ded<   dS )ÚQDQBiasQuantInfor3   r5   r4   Úweight_nameÚfloatÚbetaNr6   r.   r/   r0   rD   rD   L   s7   € € € € € € à€N€N�NØ€O€O�OØÐÐÑØ€K€K�K€K€Kr/   rD   c                  ó6   — e Zd ZU ded<   ded<   ded<   d
d„Zd	S )ÚQDQTensorQuantParamsr   ÚoriginalzQuantizationParams | NoneÚ	convertedúset[str] | NoneÚconverted_recv_nodesÚreturnc                óh   — | j         €| j        S | j        €| j         S || j        v r| j         n| j        S r;   ©rK   rJ   rM   ©rA   Úconsumer_node_names     r0   Úget_for_consumerz%QDQTensorQuantParams.get_for_consumer]   óB   € ØŒ>Ð!Ø”=Ð àÔ$Ð,Ø”>Ð!ð
 #5¸Ô8QÐ"QÐ"QˆtŒ~ˆ~ÐX\ÔXeÐer/   N)rN   r   ©r(   r)   r*   r7   rS   r.   r/   r0   rI   rI   W   sT   € € € € € € à Ð Ð Ñ Ø(Ð(Ð(Ñ(Ø)Ð)Ð)Ñ)ð
fð 
fð 
fð 
fð 
fð 
fr/   rI   c                  ó$   — e Zd ZU ded<   ded<   dS )ÚQDQScaleZpInitializersr   ÚscaleÚ
zero_pointNr6   r.   r/   r0   rW   rW   k   s*   € € € € € € àÐÐÑØÐÐÑÐÐr/   rW   c                  ó.   — e Zd ZU ded<   ded<   ded<   dS )ÚQDQTensorScaleZpInitializersrW   rJ   zQDQScaleZpInitializers | NonerK   rL   rM   Nr6   r.   r/   r0   r[   r[   t   s6   € € € € € € à$Ð$Ð$Ñ$Ø,Ð,Ð,Ñ,Ø)Ð)Ð)Ñ)Ð)Ð)r/   r[   c                  ó6   — e Zd ZU ded<   ded<   ded<   d
d„Zd	S )ÚQDQTensorQuantizedValuer   rJ   zQuantizedValue | NonerK   rL   rM   rN   c                óh   — | j         €| j        S | j        €| j         S || j        v r| j         n| j        S r;   rP   rQ   s     r0   rS   z(QDQTensorQuantizedValue.get_for_consumer„   rT   r/   N)rN   r   rU   r.   r/   r0   r]   r]   ~   sT   € € € € € € àÐÐÑØ$Ð$Ð$Ñ$Ø)Ð)Ð)Ñ)ð
fð 
fð 
fð 
fð 
fð 
fr/   r]   c                  ó0  — e Zd Z	 dYd„Zd„ Zd„ Zdej        fd„ZdZd„Z	d[d„Z
dZd„Zd„ Zd\d„Zd]d„Zd^d„Zd„ Zd„ Zd „ Zd!„ Zd"„ Z	 	 d_d`d-„Z	 	 d_dad1„Z	 	 d_dbd2„Zdcd4„ZdYd5„Zd6„ Zd7„ Zd8„ Zd9„ ZdZd:„Z	 dYddd?„ZdedB„Z dfdF„Z!	 dgdhdM„Z"didO„Z#djdT„Z$dkdV„Z%dldX„Z&dS )mÚQDQQuantizerNc                ó
  ‡— t          j        | |||||||||	|
¦  «         i | _        i | _        g | _        |
                     dg ¦  «        | _        |
                     dd¦  «        | _        |
                     dd¦  «        | _        |
                     dd¦  «        | _	        i | _
        i | _        |
                     di ¦  «        | _        |
                     dd¦  «        rt          nd | _        |
                     d	d¦  «        | _        |
                     d
d¦  «        | _        |
                     dd¦  «        | _        | j        dk     r’t&          j        t&          j        t&          j        t&          j        fŠt1          ˆfd„| j        D ¦   «         ¦  «        }| j        s=| j        ‰v s| j        ‰v s|r)t9          j        dt          › d�¦  «         t          | _        |                      ¦   «         | _        i | _         i | _!        d S )NÚ"OpTypesToExcludeOutputQuantizationÚAddQDQPairToWeightFÚQuantizeBiasTÚDedicatedQDQPairÚ QDQOpTypePerChannelSupportToAxisÚUseQDQContribOpsÚQDQKeepRemovableActivationsÚ"QDQDisableWeightAdjustForInt32BiasÚ	BlockSizer   é   c              3  ó*   •K  — | ]}|j         ‰v V — Œd S r;   )r<   )Ú.0ÚtÚopset21_typess     €r0   ú	<genexpr>z(QDQQuantizer.__init__.<locals>.<genexpr>á   s;   øè è € ð /ð /Ø34�” Ð.ð/ð /ð /ð /ð /ð /r/   zÉONNX QuantizeLinear and DequantizeLinear operators do not support 16-bit/4-bit integer quantization types prior to opset 21. The domain of QuantizeLinear and DequantizeLinear operators will be set to 'z' to enable support.)"r
   rB   Útensors_to_quantizeÚbias_to_quantizeÚnodes_to_removeÚgetÚ'op_types_to_exclude_output_quantizationÚadd_qdq_pair_to_weightÚquantize_biasÚdedicated_qdq_pairÚtensor_to_its_receiving_nodesÚtensor_to_producing_dqÚ'qdq_op_type_per_channel_support_to_axisr   Úqdq_op_domainÚqdq_keep_removable_activationsÚ(qdq_disable_weight_adjust_for_int32_biasÚ
block_sizeÚopset_versionr   ÚUINT16ÚINT16ÚUINT4ÚINT4ÚanyÚtensor_quant_override_qtypesÚactivation_qTypeÚweight_qTypeÚloggingÚwarningÚcalc_graph_quant_paramsÚquantization_paramsÚinitializer_quant_paramsÚquantized_value_map)rA   ÚmodelÚper_channelÚreduce_rangerˆ   r‡   Útensors_rangeÚnodes_to_quantizeÚnodes_to_excludeÚop_types_to_quantizeÚextra_optionsÚoverrides_have_opset21_typesro   s               @r0   rB   zQDQQuantizer.__init__’   sc  ø€ õ 	ÔØØØØØØØØØØ Øñ	
ô 	
ð 	
ð CEˆÔ Ø=?ˆÔà!ˆÔð 8E×7HÒ7HÐImÐoqÑ7rÔ7rˆÔ4ð
 '4×&7Ò&7Ð8LÈeÑ&TÔ&TˆÔ#ð +×.Ò.¨~¸tÑDÔDˆÔð #0×"3Ò"3Ð4FÈÑ"NÔ"NˆÔØNPˆÔ*ð BDˆÔ#ð 8E×7HÒ7HÐIkÐmoÑ7pÔ7pˆÔ4à*7×*;Ò*;Ð<NÐPUÑ*VÔ*VÐ`�Y˜YÐ\`ˆÔð /<×.?Ò.?Ð@]Ð_dÑ.eÔ.eˆÔ+ð 9F×8IÒ8IÐJnÐpuÑ8vÔ8vˆÔ5ð  -×0Ò0°¸aÑ@Ô@ˆŒð
 Ô Ò"Ð"Ý(Ô/µÔ1BÅKÔDUÕWbÔWgÐhˆMÝ+.ð /ð /ð /ð /Ø8<Ô8Yð/ñ /ô /ñ ,ô ,Ð(ð Ô%ð /ØÔ%¨Ð6Ð6ØÔ$¨Ð5Ð5Ø/ð 6õ ”ð&åclð&ð &ð &ñô ð õ &/�Ô"à#'×#?Ò#?Ñ#AÔ#AˆÔ ØGIˆÔ%ð $&ˆÔ Ð Ð r/   c                óè   — t          || j                             ¦   «         ¦  «        }|�|j        S || j        v r8| j        |         }|j                             d¦  «        r|j        j        j        S dS )ú2
        Check if tensor can be quantized
        Nr<   )	r   r�   Úinitializerr@   Úvalue_infosÚtypeÚHasFieldr<   Ú	elem_type©rA   Útensor_nameÚweightÚvis       r0   Ú_get_tensor_typezQDQQuantizer._get_tensor_type÷   sv   € õ ˜k¨4¬:×+AÒ+AÑ+CÔ+CÑDÔDˆØÐØÔ#Ð#Ø˜DÔ,Ð,Ð,ØÔ! +Ô.ˆBØŒw×Ò Ñ.Ô.ð 5Ø”wÔ*Ô4Ð4Øˆtr/   c                ó˜  — t          || j                             ¦   «         ¦  «        }|�,|j        t          j        j        t          j        j        fv rdS nt|| j        v rS| j        |         }|j	         
                    d¦  «        r+|j	        j        j        t
          j        t
          j        fv rdS nt          j        d|› d�¦  «         dS )r™   NTr<   z$failed to infer the type of tensor: z6. Skip to quantize it. Please check if it is expected.F)r   r�   rš   r@   Ú
onnx_protor   ÚFLOATÚFLOAT16r›   rœ   r�   r<   rž   r‰   rŠ   rŸ   s       r0   Ú_is_tensor_quantizablez#QDQQuantizer._is_tensor_quantizable  sÛ   € õ ˜k¨4¬:×+AÒ+AÑ+CÔ+CÑDÔDˆØÐØÔ¥JÔ$:Ô$@Å*ÔBXÔB`Ð#aÐaÐaØ�tð bà˜DÔ,Ð,Ð,ØÔ! +Ô.ˆBØŒw×Ò Ñ.Ô.ð °2´7Ô3FÔ3PÝÔ!ÝÔ#ðUð 4ð 4ð �tøåŒOØz°{ÐzÐzÐzñô ð ð ˆur/   c                óv  — |                       |¦  «        r¡|rft          |t          ¦  «        s t          dt	          |¦  «        › d�¦  «        ‚|                      |¦  «        }t          |||¬¦  «        | j        |<   dS || j        vr2|                      |¦  «        }t          ||¬¦  «        | j        |<   dS dS dS )a  
        Adds a tensor to the list (actually a dict) of tensors to quantize. Called indirectly by op quantizers that
        want to quantize a tensor (i.e., "mark" a tensor for quantization).

        If quant_sharing_provider is not None, tensor with name tensor_name will be quantized with the same
        quantization parameters as the node input specified in quant_sharing_provider. Ex: A Tranpose node's output
        will typically use the same quantization parameter initializers used at the Transpose node's input.

        Args:
            tensor_name: name of the tensor to quantize
            quant_sharing_provider: name of the tensor and node that provides quantization parameter
            tensor_type: QDQQuantTensorType default ACTIVATION
        zBquant_sharing_provider must be of type QDQQuantParamProvider, not ú.)r<   r=   r@   )r<   r@   N)r¨   Ú
isinstancer2   Ú	TypeErrorrœ   r£   r9   rq   )rA   r    Úquant_sharing_providerr<   r@   s        r0   Ú__quantize_tensorzQDQQuantizer.__quantize_tensor  sû   € ð ×&Ò& {Ñ3Ô3ð 	yØ%ð yÝ!Ð"8Õ:OÑPÔPð Ý#Ø|Õ]aÐbxÑ]yÔ]yÐ|Ð|Ð|ñô ð ð !×1Ò1°+Ñ>Ô>�	Ý8JØ +ÐAWÐclð9ñ 9ô 9�Ô(¨Ñ5Ð5Ð5ð  DÔ$<Ð<Ð<Ø ×1Ò1°+Ñ>Ô>�	Ý8JÐWbÐnwÐ8xÑ8xÔ8x�Ô(¨Ñ5Ð5Ð5ð	yð 	yð =Ð<r/   r    r3   c                óD   — |                       |dt          j        ¦  «        S )zó
        Adds a tensor to the list of tensors to quantize. Called by op quantizers that
        want to quantize a tensor (i.e., "mark" a tensor for quantization).

        Args:
            tensor_name: name of the tensor to quantize
        N)Ú_QDQQuantizer__quantize_tensorr&   r+   ©rA   r    s     r0   Úquantize_activation_tensorz'QDQQuantizer.quantize_activation_tensor7  s    € ð ×%Ò% k°4Õ9KÔ9VÑWÔWÐWr/   Úoutput_namer4   r5   c                ó`   — |                       |t          ||¦  «        t          j        ¦  «        S )a˜  
        Adds a tensor to the list of tensors to quantize. Called by op quantizers that
        want to quantize an output tensor using the same quantization parameters as one of the node's inputs.

        Ex: A Tranpose node's output will typically use the same quantization parameter initializers used at
        the Transpose node's input.

        Args:
            output_name: name of the node output to quantize so that it uses the same quantization params as an input.
            input_name: name of the node input from which the output tensor will get its quantization params.
            node_name: name of the node that consumes `input_name`.
        )r°   r2   r&   r+   )rA   r³   r4   r5   s       r0   Úquantize_output_same_as_inputz*QDQQuantizer.quantize_output_same_as_inputA  s2   € ð ×%Ò%ØÕ.¨z¸9ÑEÔEÕGYÔGdñ
ô 
ð 	
r/   c                óD   — |                       |dt          j        ¦  «        S )zú
        Adds a tensor to the list of weight tensors to quantize. Called by op quantizers that
        want to quantize a weight (i.e., "mark" a weight for quantization).

        Args:
            tensor_name: name of the weight to quantize
        N)r°   r&   r,   r±   s     r0   Úquantize_weight_tensorz#QDQQuantizer.quantize_weight_tensorR  s    € ð ×%Ò% k°4Õ9KÔ9RÑSÔSÐSr/   c                ó4  — t          || j                             ¦   «         ¦  «        }|rV|j        t          j        j        t          j        j        fv r+t          t          j
        ||j        ¬¦  «        | j        |<   d S d S t          j        d|› d�¦  «         d S )N)r<   r>   r@   z9only support per-channel quantization on weight. Tensor: z is not quantized.)r   r�   rš   r@   r¥   r   r¦   r§   r9   r&   r,   rq   r‰   rŠ   )rA   r    r>   r¡   s       r0   Ú"quantize_weight_tensor_per_channelz/QDQQuantizer.quantize_weight_tensor_per_channel\  s¥   € Ý˜k¨4¬:×+AÒ+AÑ+CÔ+CÑDÔDˆØð 	yØÔ¥JÔ$:Ô$@Å*ÔBXÔB`Ð#aÐaÐaÝ8JÝ 2Ô 9ÀÐPVÔP`ð9ñ 9ô 9�Ô(¨Ñ5Ð5Ð5ð bÐaõ
 ŒOÐwÐXcÐwÐwÐwÑxÔxÐxÐxÐxr/   rš   úonnx.TensorProtorN   c                óò   — | j                              |j        ¦  «        dz   }|j        › |› �}t          j        ¦   «         }|                     |¦  «         ||_        | j                              |¦  «         |S )zk
        Duplicates an existing initializer and adds it to the model. Returns the new initializer.
        r	   )r�   Ú#get_largest_initializer_name_suffixÚnameÚonnxr   ÚCopyFromÚadd_initializer)rA   rš   Úname_suffixÚnew_initializer_nameÚnew_initializers        r0   Ú_dup_initializerzQDQQuantizer._dup_initializerf  s|   € ð  œ:×IÒIÈ+ÔJZÑ[Ô[Ð^_Ñ_ˆØ"-Ô"2ÐA°KÐAÐAÐÝÔ*Ñ,Ô,ˆØ× Ò  Ñ-Ô-Ð-Ø3ˆÔØŒ
×"Ò" ?Ñ3Ô3Ð3ØÐr/   ç      ð?c                óü  — | j                              |¦  «        rbt          j        d|› d�¦  «         |                      |d¬¦  «        \  }}|r|                      ||¦  «         n|                      |¦  «         dS t          || j         	                    ¦   «         ¦  «        }|€t          j
        d|› d�¦  «         dS |j        t          j        j        t          j        j        fvrt          j        d|› d�¦  «         dS |}	|| j        v rT|                      |¦  «        }
|
j        }	| j                             ||	|h¦  «         t          j        d	|› d
|	› d�¦  «         t)          ||||¦  «        | j        |	<   dS )a¾  
        Adds a bias tensor to the list of bias tensors to quantize. Called by op quantizers that
        want to quantize a bias with bias_zero_point = 0 and bias_scale = input_scale * weight_scale * beta.
        TODO: Explain the reasoning for using this formula.

        Args:
            node_name: name of the node that consumes the bias, input, and weight tensors.
            bias_name: name of the bias tensor to quantize.
            input_name: name of the input tensor whose scale is used to compute the bias's scale.
            weight_name: name of the weight tensor whose scale is used to compute the bias's scale.
            beta: Multiplier used to compute the bias's scale.
        zQuantizing bias tensor 'z=' as a weight due to the presence of user-specified overridesr   )Údefault_axisNzExpected bias 'z' to be an initializerz%' to be an floating-point initializerzCreated a copy of bias input 'z
' called 'ú')Útensor_quant_overridesrt   r‰   ÚinfoÚis_tensor_per_channelr¹   r·   r   r�   rš   rŠ   r@   r¥   r   r¦   r§   rr   rÄ   r½   Úreplace_input_of_nodesrD   )rA   r5   Ú	bias_namer4   rE   rG   Úis_per_channelr>   Úbias_initializerÚactual_bias_nameÚnew_bias_initializers              r0   Úquantize_bias_tensorz!QDQQuantizer.quantize_bias_tensorr  s½  € ð Ô&×*Ò*¨9Ñ5Ô5ð 		ÝŒLØs¨9ÐsÐsÐsñô ð ð $(×#=Ò#=¸iÐVWÐ#=Ñ#XÔ#XÑ ˆN˜DØð 7Ø×7Ò7¸	À4ÑHÔHÐHÐHà×+Ò+¨IÑ6Ô6Ð6ØˆFå'¨	°4´:×3IÒ3IÑ3KÔ3KÑLÔLÐØÐ#ÝŒOÐO¨iÐOÐOÐOÑPÔPÐPØˆFàÔ%­jÔ.DÔ.JÍJÔLbÔLjÐ-kÐkÐkÝŒLÐ[¨9Ð[Ð[Ð[Ñ\Ô\Ð\ØˆFà$ÐØ˜Ô-Ð-Ð-ð $(×#8Ò#8Ð9IÑ#JÔ#JÐ Ø3Ô8Ðð ŒJ×-Ò-¨iÐ9IÈIÈ;ÑWÔWÐWÝŒLÐb¸)ÐbÐbÐO_ÐbÐbÐbÑcÔcÐcõ 3CÀ9ÈjÐZeÐgkÑ2lÔ2lˆÔÐ.Ñ/Ð/Ð/r/   Úinput_scaleú
np.ndarrayÚweight_scalerE   Úbias_tprÎ   Úboolútuple[bool, np.ndarray | None]c                ó”  — |j         sdS t          |¦  «        }t          j        t          j        ¦  «        }d}t          j        |j        t          j        ¬¦  «        t          j        |j        dz   t          j        ¬¦  «        z
  }	|j	        }
d}|�s‰t          j
        |                     ¦   «         t          j        dt          j        ¬¦  «        ¦  «        }t          j        |                     ¦   «         t          j        dt          j        ¬¦  «        ¦  «        }t          j        t          j        |¦  «        t          j        |¦  «        ¦  «        }|d|z  z  |	z  }t          j        |                     ¦   «         t          j        ¬¦  «        }t          j        |                     ¦   «         t          j        ¬¦  «        }||z  }||k     rJ|dk    rD||z  }t          j        d	|› d
|› d|j        › d�¦  «         ||z  }|                     |
¦  «        }d}�n*|j        �r"t'          |j        ¦  «        dk    �r	|j        d         }t)          |¦  «        D ]ì}t          j        ||         ¦  «        }|d|z  z  |	z  }t          j        |                     ¦   «         t          j        ¬¦  «        }t          j        ||                              ¦   «         t          j        ¬¦  «        }||z  }||k     rP|dk    rJ||z  }t          j        d|› d|› d|› d|j        › d�	¦  «         ||z  }|                     |
¦  «        ||<   d}Œí||fS )aI  
        Checks if the bias scale (input_scale * weight_scale) that we intend to use is too small.
        A bias scale that is too small leads to quantized bias values that fall outside the range of a int32 and have to
        be clipped, which decreases accuracy. If this function detects such a scenario, the weight_scale value will be
        increased to prevent this from happening.

        Although the adjustment method and amount differs, the idea to adjust the weight's scale came from the following
        reference:
        https://github.com/tensorflow/tensorflow/blob/master/tensorflow/lite/tools/optimize/quantization_utils.cc#L252

        :param input_scale: The input's scale.
        :param weight_scale: The weight scale to potentially adjust.
        :param weight_name: The weight initializer's name. Used for logging.
        :param bias_tp: The bias ONNX initializer.
        :param is_per_channel: True if the bias and weight are quantized per-channel.
        :return: A tuple with a bool indicating if the weight's scale was adjusted and the new weight scale.
        ©FNgq¬‹Ûh ð?©Údtyper	   Fr   g       @g        zIncreasing scale for weight `z` by the ratio z to ensure bias input `z` has a valid scale.TzIncreased scale[z] for weight `z` by ratio )Úsizer#   ÚnpÚiinfoÚint32ÚarrayÚmaxÚfloat64ÚminrÜ   ÚminimumÚmaximumÚabsÚitemr‰   rÊ   r½   ÚastypeÚshapeÚlenÚrange)rA   rÓ   rÕ   rE   rÖ   rÎ   Úbias_float_dataÚ
int32_infoÚmultiplicative_epsilonÚqrangeÚweight_scale_dtypeÚupdated_an_elemÚrminÚrmaxÚabsmaxÚbias_smallest_valid_scaleÚinput_scale_fp64Úweight_scale_fp64Úbias_candidate_scaleÚratioÚ	new_scaleÚ	num_elemsÚiÚ	bias_rmaxs                           r0   Ú#_adjust_weight_scale_for_int32_biasz0QDQQuantizer._adjust_weight_scale_for_int32_bias£  sx  € ð2 Ô ð 	Ø�;å/°Ñ8Ô8ˆå”X�bœhÑ'Ô'ˆ
Ø!'ÐÝ”˜*œ.µ´
Ð;Ñ;Ô;½b¼hÀzÄ~ÐXYÑGYÕacÔakÐ>lÑ>lÔ>lÑlˆØ)Ô/ÐØˆàñ (	+Ý”:˜o×1Ò1Ñ3Ô3µR´X¸aÅrÄzÐ5RÑ5RÔ5RÑSÔSˆDÝ”:˜o×1Ò1Ñ3Ô3µR´X¸aÅrÄzÐ5RÑ5RÔ5RÑSÔSˆDÝ”Z¥¤ t¡¤­b¬f°T©l¬lÑ;Ô;ˆFØ(>À#ÈÁ,Ñ(OÐRXÑ(XÐ%å!œx¨×(8Ò(8Ñ(:Ô(:Å"Ä*ÐMÑMÔMÐÝ "¤¨×):Ò):Ñ)<Ô)<ÅBÄJÐ OÑ OÔ OÐØ#3Ð6GÑ#GÐ à$Ð'@Ò@Ð@ÐG[Ð^aÒGaÐGaà1Ð4HÑH�Ý”ðM°Kð Mð MÐPUð Mð MØ*1¬,ðMð Mð Mñô ð ð .°Ñ5�	Ø(×/Ò/Ð0BÑCÔC�Ø"&�ùØÔñ 	+¥C¨Ô(:Ñ$;Ô$;¸qÒ$@Ñ$@à$Ô*¨1Ô-ˆIå˜9Ñ%Ô%ð +ð +�ÝœF ?°1Ô#5Ñ6Ô6�	Ø,BÀcÈIÁoÑ,VÐY_Ñ,_Ð)å#%¤8¨K×,<Ò,<Ñ,>Ô,>ÅbÄjÐ#QÑ#QÔ#QÐ Ý$&¤H¨\¸!¬_×-AÒ-AÑ-CÔ-CÍ2Ì:Ð$VÑ$VÔ$VÐ!Ø'7Ð:KÑ'KÐ$Ø(Ð+DÒDÐDÐK_ÐbeÒKeÐKeà5Ð8LÑL�EÝ”LðT¨1ð Tð T¸Kð Tð TÐTYð Tð TØ18´ðTð Tð Tñô ð ð !2°EÑ 9�IØ&/×&6Ò&6Ð7IÑ&JÔ&J�L ‘OØ&*�Oøà Ð,Ð,r/   c                ó8  — | j         rdS | j                             ¦   «         D �]u\  }}|j        | j        vs|j        | j        vs|j        | j        vrŒ1| j        |j                                      |j	        ¦  «        }| j        |j                 }t          j        |d         t          j                             |j        ¦  «        ¬¦  «        }| j        |j                 }|d         }|t          j        j        t          j        j        fvrŒê|d         }|                     ¦   «         r�Œ|d         }	|                     dd¦  «        du}
|                      ||	|j        t-          || j                             ¦   «         ¦  «        |
¦  «        \  }}|r||d<   �ŒwdS )a3  
        Iterates through all bias inputs that should be quantized to int32. If the intended
        bias scale (equal to input_scale * weight_scale) is too small, this function will increase
        the associated weight's scale to ensure the bias does not overflow the int32 range when quantized.
        NrX   rÛ   Ú
quant_typerY   r>   )r~   rr   Úitemsr4   rŒ   rq   rE   r�   rS   r5   rÞ   Úasarrayr¾   ÚhelperÚtensor_dtype_to_np_dtyper@   r   ÚINT8r‚   r…   rt   rÿ   r   r�   rš   )rA   rÍ   Ú	bias_infoÚinput_qparamsÚ
input_inforÓ   Úweight_quant_paramsÚweight_quant_typeÚweight_zero_pointrÕ   rÎ   Údid_update_weight_scaleÚnew_weight_scales                r0   Ú,_adjust_weight_quant_params_for_bias_tensorsz9QDQQuantizer._adjust_weight_quant_params_for_bias_tensorsó  sÆ  € ð Ô8ð 	àˆFà$(Ô$9×$?Ò$?Ñ$AÔ$Að &	@ñ &	@Ñ ˆI�yàÔ$¨DÔ,DÐDÐDØÔ'¨tÔ/GÐGÐGØÔ(°Ô0MÐMÐMàð !Ô4°YÔ5IÔJ×[Ò[Ð\eÔ\oÑpÔpˆMØÔ1°)Ô2FÔGˆJÝœ*Ø˜gÔ&­d¬k×.RÒ.RÐS]ÔSgÑ.hÔ.hðñ ô ˆKð #'Ô"?À	Ô@UÔ"VÐØ 3°LÔ AÐØ ­Ô)9Ô)>ÅÔ@PÔ@VÐ(WÐWÐWØà,?ÀÔ,MÐØ ×$Ò$Ñ&Ô&ð áà':¸7Ô'CˆLØ0×4Ò4°V¸TÑBÔBÈ$ÐNˆNð 9=×8`Ò8`ØØØÔ%Ý˜Y¨¬
×(>Ò(>Ñ(@Ô(@ÑAÔAØñ9ô 9Ñ5Ð#Ð%5ð 'ð @Ø/?Ð# GÑ,ùðM&	@ð &	@r/   c                ó:   — | j                              |¦  «         d S r;   )rs   Úappend)rA   Únodes     r0   Úremove_nodezQDQQuantizer.remove_node&  s   € ØÔ×#Ò# DÑ)Ô)Ð)Ð)Ð)r/   c                óD   — | j                              | j        ¦  «         d S r;   )r�   Úremove_nodesrs   )rA   s    r0   r  zQDQQuantizer.remove_nodes)  s!   € ØŒ
×Ò Ô 4Ñ5Ô5Ð5Ð5Ð5r/   c                ó†  — | j                              ¦   «         D ]œ}|                      |¦  «        rat          | |¦  «        }|                     ¦   «          |j        D ]5}|| j        vr
g | j        |<   | j        |                              |¦  «         Œ6|j        t          k    r|j
        D ]}|| j        |<   ŒŒ�|                      ¦   «         | _        |                      ¦   «          |                      ¦   «          |                      ¦   «          | j        r|                      ¦   «          |                      ¦   «          | j        s| j                              ¦   «          t,          | j         j         _        t0          | j         j         _        | j        t6          k    r | j                              t6          d¦  «         | j         j         S )Nr	   )r�   ÚnodesÚshould_quantize_noder$   ÚquantizeÚinputry   r  Úop_typer   Úoutputrz   Ú_calc_initializer_quant_paramsr�   r  Ú_quantize_normal_tensorsÚ_quantize_sharing_param_tensorsrw   Ú_quantize_bias_tensorsr  rv   Úclean_initializersr   Úproducer_namer   Úproducer_versionr|   r   Úset_opset_import)rA   r  Úop_quantizerr    s       r0   Úquantize_modelzQDQQuantizer.quantize_model,  sº  € Ø”J×$Ò$Ñ&Ô&ð 	Dð 	DˆDØ×(Ò(¨Ñ.Ô.ð QÝ1°$¸Ñ=Ô=�Ø×%Ò%Ñ'Ô'Ð'à#'¤:ð Qð Q�KØ"¨$Ô*LÐLÐLØJL˜Ô:¸;ÑGØÔ6°{ÔC×JÒJÈ4ÑPÔPÐPÐPØŒ|�Ò.Ð.Ø#'¤;ð Dð D�KØ?C�DÔ/°Ñ<Ð<øà(,×(KÒ(KÑ(MÔ(MˆÔ%Ø×9Ò9Ñ;Ô;Ð;Ø×%Ò%Ñ'Ô'Ð'Ø×,Ò,Ñ.Ô.Ð.ØÔð 	*Ø×'Ò'Ñ)Ô)Ð)Ø×ÒÑÔÐØÔ*ð 	,ØŒJ×)Ò)Ñ+Ô+Ð+å)5ˆŒ
ÔÔ&Ý,7ˆŒ
ÔÔ)ØÔ¥Ò*Ð*ØŒJ×'Ò'­	°1Ñ5Ô5Ð5àŒzÔÐr/   c                ó„  — || j         v r¶| j         |         j        €¤| j         |         j        €’t          | j                             ¦   «         |         ¦  «        dk    rb| j                             |¦  «        sH| j                             |¦  «        s.| j                             ||¦  «         || j        v r| j        |= dS dS )Nr	   TF)	rŒ   rK   rë   r�   Úinput_name_to_nodesÚis_graph_outputÚis_graph_inputÚreplace_output_of_all_nodesrq   )rA   Úupstream_output_namer³   s      r0   Útry_replacing_upstream_outputz*QDQQuantizer.try_replacing_upstream_outputK  sÍ   € à˜4Ô3Ð3Ð3ØÔ(¨Ô5Ô?ÐGØÔ(Ð)=Ô>ÔHÐPÝ�D”J×2Ò2Ñ4Ô4Ð5IÔJÑKÔKÈqÒPÐPØ”J×.Ò.Ð/CÑDÔDð Qà”J×-Ò-Ð.BÑCÔCð Qð ŒJ×2Ò2Ð3GÈÑUÔUÐUØ# tÔ'?Ð?Ð?ØÔ,Ð-AÐBØ�4Øˆur/   r   Úq_inputÚq_outputÚquant_node_nameÚ
scale_nameÚzp_namer>   ú
int | Noner   Úintc                ó¤   — || j         dœ}|r||d<   t          j        j        t          |||g|g|fi |¤Ž}	| j                             |	g¦  «         dS )zI
        Creates a QuantizeLinear node and adds it to the model.
        ©r>   Údomainr   N)r|   r¾   r  Ú	make_noder   r�   Ú	add_nodes)
rA   r.  r/  r0  r1  r2  r>   r   ÚkwargsÚqlinear_nodes
             r0   Ú_create_q_nodezQDQQuantizer._create_q_nodeZ  s~   € ð +/¸$Ô:LÐ!MÐ!MˆØð 	.Ø#-ˆF�<Ñ Ý”{Ô,ÝØ�j 'Ð*ØˆJØð	
ð 
ð
 ð
ð 
ˆð 	Œ
×Ò˜l˜^Ñ,Ô,Ð,Ð,Ð,r/   Údq_inputÚ	dq_outputÚdequant_node_namec                ó¤   — || j         dœ}|r||d<   t          j        j        t          |||g|g|fi |¤Ž}	| j                             |	g¦  «         dS )zK
        Creates a DequantizeLinear node and adds it to the model.
        r6  r   N)r|   r¾   r  r8  r   r�   r9  )
rA   r=  r>  r?  r1  r2  r>   r   r:  Údequant_nodes
             r0   Ú_create_dq_nodezQDQQuantizer._create_dq_nodes  s~   € ð +/¸$Ô:LÐ!MÐ!MˆØð 	.Ø#-ˆF�<Ñ Ý”{Ô,ÝØ�z 7Ð+ØˆKØð	
ð 
ð
 ð
ð 
ˆð 	Œ
×Ò˜l˜^Ñ,Ô,Ð,Ð,Ð,r/   c                óì   — |	| j         dœ}|
r|
|d<   t          j        j        t          |||g|g|fi |¤Ž}t          j        j        t
          |||g|g|fi |¤Ž}| j                             ||g¦  «         d S )Nr6  r   )r|   r¾   r  r8  r   r   r�   r9  )rA   r.  r/  r0  r=  r>  r?  r1  r2  r>   r   r:  r;  rA  s                 r0   Ú_create_qdq_nodeszQDQQuantizer._create_qdq_nodesŒ  s»   € ð +/¸$Ô:LÐ!MÐ!MˆØð 	.Ø#-ˆF�<Ñ Ý”{Ô,ÝØ�j 'Ð*ØˆJØð	
ð 
ð
 ð
ð 
ˆõ ”{Ô,ÝØ�z 7Ð+ØˆKØð	
ð 
ð
 ð
ð 
ˆð 	Œ
×Ò˜l¨LÐ9Ñ:Ô:Ð:Ð:Ð:r/   Úweight_protoc                ó’  — |j         }|| j        v rdS | j        |         }|                     d¦  «        }|                     dd¦  «        }|                      ||¦  «        }d}t          |¦  «        }| j                             ||¦  «         | j        r]t          |¦  «        }	|  
                    ||	t          |¦  «        |	|t          |¦  «        |j        j         |j        j         ||¬¦
  «
         nŠt          ||d         |d         |d         ||¬¦  «        }
| j                             |
¦  «         |
j         }|                      |
j         |t          |¦  «        |j        j         |j        j         ||¬	¦  «         t%          |||j        j         |j        j         t&          j        |¬
¦  «        }t+          |dd¦  «        | j        |<   dS )a  
        Adds Q/DQ nodes for an initializer. If `self.add_qdq_pair_to_weight` is true, creates
        the sequence (weight_f32 -> Q -> DQ -> ). Otherwise, this function quantizes the initializer
        and adds the sequence (weight_quant -> DQ ->).
        Nr>   r   r   )r   r  rY   rX   )r>   r   )r>   )r½   rŽ   r�   rt   Ú_make_scale_zp_initializersr   r�   Úreplace_input_of_all_nodesrv   r   rD  r   r   rX   rY   r!   rÀ   rB  r   r   ÚInitializerr]   )rA   rE  rE   Úquant_paramsr>   r   Úscale_zp_initializersÚq_weight_nameÚweight_dequant_outputÚweight_quant_outputÚquant_weightÚquantized_values               r0   Ú_add_qdq_nodes_for_initializerz+QDQQuantizer._add_qdq_nodes_for_initializer¬  s  € ð #Ô'ˆØ˜$Ô2Ð2Ð2ØˆFà+/Ô+HÈÔ+UˆØ ×$Ò$ VÑ,Ô,ˆØ&×*Ò*¨<¸Ñ;Ô;ˆ
Ø $× @Ò @ÀÈlÑ [Ô [ÐØ$(ˆÝ 9¸+Ñ FÔ FÐØŒ
×-Ò-¨kÐ;PÑQÔQÐQàÔ&ð '	õ #:¸+Ñ"FÔ"FÐà×"Ò"ØØ#Ý  Ñ-Ô-Ø#Ø%Ý" ;Ñ/Ô/Ø%Ô+Ô0Ø%Ô0Ô5ØØ%ð #ñ ô ð ð õ 5ØØ˜\Ô*Ø˜\Ô*Ø˜WÔ%ØØ%ðñ ô ˆLð ŒJ×&Ò& |Ñ4Ô4Ð4à(Ô-ˆMØ× Ò ØÔ!Ø%Ý" ;Ñ/Ô/Ø%Ô+Ô0Ø%Ô0Ô5ØØ%ð !ñ ô ð õ )ØØØ!Ô'Ô,Ø!Ô,Ô1ÝÔ*Øð
ñ 
ô 
ˆõ 1HÈÐY]Ð_cÑ0dÔ0dˆÔ  Ñ-Ð-Ð-r/   c                ól  — | j         �r0|| j        v �r&t          | j        |         ¦  «        dk    �rt          | j        |         ¦  «        }t          |¦  «        D ]Û}d|dz   › �}t	          |¦  «        |z   }t          |¦  «        |z   }	t          |¦  «        |z   }
t          |¦  «        |z   }|                      |||
||	|||¦  «         | j        |         |         }| j	         
                    |||	¦  «         |dk    r8t          ||	||t          j        |¬¦  «        }t          |d d ¦  «        | j        |<   ŒÜd S |}t          |¦  «        }| j	                             |¦  «        r-t#          |¦  «        }|}| j	                             ||¦  «         n| j	                             ||¦  «         |                      |t	          |¦  «        t          |¦  «        t	          |¦  «        |t          |¦  «        ||¦  «         t          ||||t          j        |¬¦  «        }t          |d d ¦  «        | j        |<   d S )Nr	   Ú_r   ©Ú
scale_type)rx   ry   rë   rì   r   r   r   r   rD  r�   Úreplace_node_inputr   r   ÚInputr]   rŽ   r)  r   r+  rH  )rA   r    r1  r2  r@   Únum_dedicated_qdq_pairrý   ÚpostfixÚ tensor_name_quant_output_postfixÚ"tensor_name_dequant_output_postfixÚquant_node_name_postfixÚdequant_node_name_postfixr  rP  r.  r>  s                   r0   Ú_add_qdq_pair_for_activationz)QDQQuantizer._add_qdq_pair_for_activationò  s“  € àÔ#ñ@	ià˜tÔAÐAÑAÝ�DÔ6°{ÔCÑDÔDÀqÒHÑHå%(¨Ô)KÈKÔ)XÑ%YÔ%YÐ"ÝÐ1Ñ2Ô2ð qð q�Ø%˜a !™e˜+˜+�Ý3JÈ;Ñ3WÔ3WÐZaÑ3aÐ0Ý5NÈ{Ñ5[Ô5[Ð^eÑ5eÐ2Ý*:¸;Ñ*GÔ*GÈ'Ñ*QÐ'Ý,>¸{Ñ,KÔ,KÈgÑ,UÐ)Ø×&Ò&ØØ4Ø+Ø4Ø6Ø-ØØñ	ô 	ð 	ð Ô9¸+ÔFÀqÔI�Ø”
×-Ò-¨d°KÐAcÑdÔdÐdØ˜’6�6Ý&4Ø#Ø:Ø"ØÝ*Ô0Ø#,ð'ñ 'ô '�Oõ =TÐTcÐeiÐkoÑ<pÔ<p�DÔ,¨[Ñ9øð9qð qð< "ˆGÝ1°+Ñ>Ô>ˆIØŒz×)Ò)¨+Ñ6Ô6ð NÝ0°Ñ=Ô=�Ø'�	Ø”
×6Ò6°{ÀGÑLÔLÐLÐLà”
×5Ò5°kÀ9ÑMÔMÐMà×"Ò"ØÝ'¨Ñ4Ô4Ý  Ñ-Ô-Ý'¨Ñ4Ô4ØÝ" ;Ñ/Ô/ØØñ	ô 	ð 	õ -ØØØØÝ"Ô(Ø$ðñ ô ˆOõ 5LÈOÐ]aÐcgÑ4hÔ4hˆDÔ$ [Ñ1Ð1Ð1r/   c                ób  — d„ | j                              |g ¦  «        D ¦   «         }| j        r6|| j         v r-t          | j         |         ¦  «        dk    rt	          d¦  «        ‚|}	|€|}t          ¦   «         }	n|	|z
  }	t          |¦  «        t          |¦  «        k    }
| j                             |¦  «        }|}|r*t          |¦  «        }| j         	                    ||¦  «         t          |¦  «        }|                      ||t          |¦  «        ||¦  «         t          |¦  «        }|r|
s|}|	r"||k    r| j                             |||	¦  «         |                      ||t!          |¦  «        ||¦  «         |}|
s;t          |› d�¦  «        }|                      ||t!          |› d�¦  «        ||¦  «         t          |› d�¦  «        }|                      ||t          |› d�¦  «        ||¦  «         t          |› d�¦  «        }|r|
r|}|r"||k    r| j                             |||¦  «         |                      ||t!          |› d�¦  «        ||¦  «         t#          ||||t$          j        |¬¦  «        }t#          ||||t$          j        |¬¦  «        }t)          |||¦  «        | j        |<   dS )a†  
        Adds Q and DQ ops to a tensor whose quantized data type is converted. That is, some consumers may use the
        original data type from the producer, while other consumers use the converted data type.
        This is generally done by adding a sequence of ops that convert from one data type (e.g., uint8) to another (e.g., uint16).

        T_float ---> Quant(to u8) ---> Convert(to u16) ---> Dequant(to float) ---> T_float'
        where Convert(to u16) is equivalent to: ---> Dequant(to float) ---> Quant(to u16) --->

        This function handles the following scenarios:

        1) Tensor T is not a graph output; all consumers use the converted type

            <Producer> ---> Q1 ---> DQ1 ---> Q2 ---> DQ2 ---> <Consumers>

        2) Tensor T is not a graph output; some consumers use the original type, others use the converted type

            <Producer> ---> Q1 -+-> DQ1 ---> <Consumers of original type>
                                |
                                +-> DQ1' ---> Q2 ---> DQ2 ---> <Consumers of converted type>

        3) Tensor T is a graph output; all consumers use the converted type

            <Producer> ---> Q1 ---> DQ1 ---> Q2 ---> DQ2 -+-> <Consumers>
                                                          |
                                                          +-> <Graph output>

        4) Tensor T is a graph output; some consumers use the original type, others use the converted type

            <Producer> ---> Q1 -+-> DQ1 -+-> <Consumers of original type>
                                |        |
                                |        +-> <Graph output>
                                |
                                +-> DQ1' ---> Q2 ---> DQ2 ---> <Consumers of converted type>

        5) Tensor T is a graph output that is not consumed by any other nodes.

            <Producer> ---> Q1 ---> DQ1 ---> Q2 ---> DQ2 ---> <Graph output>
        c                ó   — h | ]	}|j         ’Œ
S r.   )r½   )rm   r  s     r0   ú	<setcomp>zEQDQQuantizer._add_qdq_ops_for_converted_activation.<locals>.<setcomp>e  s   € ÐkÐkÐk¨4˜TœYÐkÐkÐkr/   r	   z|Do not currently support converted quant_types in TensorQuantOverrides when the `dedicated_qdq_pair` extra_option is enabledNÚ_convertÚ_convert_clonerT  )ry   rt   rx   rë   Ú
ValueErrorÚsetr�   r)  r   r+  r   r<  r   r   rÌ   rB  r   r   r   rW  r]   rŽ   )rA   r    Úfirst_scale_nameÚfirst_zp_nameÚscale_data_typeÚconvert_scale_nameÚconvert_zp_nameÚconvert_recv_nodesÚtensor_recv_nodesÚoriginal_recv_nodesÚall_use_convertedr)  Úfirst_q_inputÚfirst_q_outputÚfirst_dq_outputÚsecond_q_inputÚsecond_q_outputÚsecond_dq_outputÚoriginal_quantized_valueÚconverted_quantized_values                       r0   Ú%_add_qdq_ops_for_converted_activationz2QDQQuantizer._add_qdq_ops_for_converted_activation5  s“  € ð` lÐk°4Ô3U×3YÒ3YÐZeÐgiÑ3jÔ3jÐkÑkÔkÐð Ô#ð	à˜tÔAÐAÐAÝ�DÔ6°{ÔCÑDÔDÀqÒHÐHõ ð Oñô ð ð 0ÐØÐ%Ø!2ÐÝ"%¡%¤%ÐÐà"5Ð8JÑ"JÐåÐ 2Ñ3Ô3µsÐ;LÑ7MÔ7MÒMÐØœ*×4Ò4°[ÑAÔAˆð $ˆØð 	OÝ2°;Ñ?Ô?ˆMØŒJ×2Ò2°;ÀÑNÔNÐNå0°Ñ=Ô=ˆØ×ÒØ˜>Õ+;¸KÑ+HÔ+HÐJZÐ\iñ	
ô 	
ð 	
õ
 4°KÑ@Ô@ˆØð 	*Ð#4ð 	*Ø)ˆOØð 	a ?°kÒ#AÐ#AØŒJ×-Ò-¨k¸?ÐL_Ñ`Ô`Ð`à×ÒØ˜OÕ-?ÀÑ-LÔ-LÐN^Ð`mñ	
ô 	
ð 	
ð )ˆØ ð 	Ý3°{Ð4LÐ4LÐ4LÑMÔMˆNØ× Ò ØØÝ" kÐ#AÐ#AÐ#AÑBÔBØ Øñô ð õ 2°[Ð2JÐ2JÐ2JÑKÔKˆØ×ÒØØÝ Ð5Ð5Ð5Ñ6Ô6ØØñ	
ô 	
ð 	
õ 5¸Ð5MÐ5MÐ5MÑNÔNÐØð 	+Ð0ð 	+Ø*ÐØð 	aÐ"2°kÒ"AÐ"AØŒJ×-Ò-¨kÐ;KÐM_Ñ`Ô`Ð`Ø×ÒØØÝ +Ð7Ð7Ð7Ñ8Ô8ØØñ	
ô 	
ð 	
õ $2ØØØØÝÔ$Ø&ð$
ñ $
ô $
Ð õ %3ØØØØÝÔ$Ø&ð%
ñ %
ô %
Ð!õ 1HØ$Ð&?ÐASñ1
ô 1
ˆÔ  Ñ-Ð-Ð-r/   c           
     ó  — | j                              ¦   «                              ¦   «         D �]\\  }}|| j        v rŒ|j        �sDt          || j                             ¦   «         ¦  «        }|r|                      |¦  «         ný|| j	        v r	| j         |= Œi|  
                    |¦  «        }|st          d|› d�¦  «        ‚|j        €=|                      ||j        j        j        |j        j        j        |j        ¬¦  «         n}|j        |j        j        j        k    sJ ‚|                      ||j        j        j        |j        j        j        |j        |j        j        j        |j        j        j        |j        ¦  «         | j         |= �Œ^dS )z}
        Adds Q/DQ ops to tensors (activations and weights) that have been marked for quantization by op quantizers.
        z4Quantization parameters are not specified for param zb. In static mode quantization params for inputs and outputs of nodes to be quantized are required.N)r@   )rq   Úcopyr  rŽ   r?   r   r�   rš   rQ  rz   Ú"_make_tensor_scale_zp_initializersrd  rK   r^  rJ   rX   r½   rY   r@   rw  rM   )rA   r    Útensor_inforš   Útensor_qparam_initializerss        r0   r  z%QDQQuantizer._quantize_normal_tensorsÒ  sÑ  € ð )-Ô(@×(EÒ(EÑ(GÔ(G×(MÒ(MÑ(OÔ(Oð 0	:ñ 0	:Ñ$ˆK˜Ø˜dÔ6Ð6Ð6ØàÔ(ñ ,:å*¨;¸¼
×8NÒ8NÑ8PÔ8PÑQÔQ�Øð 'Ø×7Ò7¸ÑDÔDÐDÐDð # dÔ&AÐAÐAØ Ô4°[ÐAØ à15×1XÒ1XÐYdÑ1eÔ1eÐ.Ø5ð Ý(ðÐS^ð ð ð ñô ð ð
 2Ô;ÐCà×9Ò9Ø'Ø6Ô?ÔEÔJØ6Ô?ÔJÔOØ&1Ô&;ð	 :ñ ô ð ð ð  +Ô4Ð8RÔ8[Ô8aÔ8kÒkÐkÐkÐkØ×BÒBØ'Ø6Ô?ÔEÔJØ6Ô?ÔJÔOØ'Ô1Ø6Ô@ÔFÔKØ6Ô@ÔKÔPØ6ÔKñô ð ð Ô,¨[Ð9ùða0	:ð 0	:r/   c           
     óü  — | j         �rs| j                              ¦   «                              ¦   «         D �]<\  }}|j        }|�r,|j        | j        v �r| j         |= | j        |j                                      |j        ¦  «        }|                      |¦  «        rt          d¦  «        ‚|| j
        v rt          d|› d�¦  «        ‚d}d}|| j        v r7| j        |         }|j        r#|                      ||j        d¦  «        }|j        }|€"|                      ||j        |j        ¦  «         Œù|                      ||j        |j        |j        j        |j        j        |j        j        |¦  «         �Œ>| j         �°qdS dS )a{  
        Adds Q/DQ ops to tensors that have been marked for quantization by op quantizers.
        Only operates on tensors that want to use the quantization parameter initializers from an upstream tensor.
        For example, a Transpose node's output tensor will typically want to use the same quantization parameter
        initializers as the Transpose node's input.
        zBQuantization parameter shared mode is not supported for weight yetz5Quantization parameter sharing is invalid for tensor z& because it has already been quantizedNrb  )rq   ry  r  r=   r4   rŽ   rS   r5   Úis_input_a_initializerrd  rz   rŒ   rK   rG  rM   r^  r1  r2  rw  rX   r@   r½   rY   )rA   r    r{  Úquant_providerrP  Úconverted_qparam_initsrM   Útensor_paramss           r0   r  z,QDQQuantizer._quantize_sharing_param_tensors  sí  € ð Ô&ñ /	Ø,0Ô,D×,IÒ,IÑ,KÔ,K×,QÒ,QÑ,SÔ,Sð .ñ .Ñ(�˜[Ø!,Ô!@�Ø!ñ , nÔ&?À4ÔC[Ð&[Ñ&[ØÔ0°Ð=à&*Ô&>¸~Ô?XÔ&Y×&jÒ&jØ&Ô0ñ'ô '�Oð ×2Ò2°;Ñ?Ô?ð oÝ(Ð)mÑnÔnÐnà" dÔ&AÐAÐAÝ(ðDÐT_ð Dð Dð Dñô ð ð .2Ð*Ø+/Ð(Ø" dÔ&>Ð>Ð>Ø(,Ô(@ÀÔ(M˜Ø(Ô2ð VØ59×5UÒ5UØ +¨]Ô-DÀjñ6ô 6Ð2ð 4AÔ3UÐ0à-Ð5à×9Ò9Ø'¨Ô)CÀ_ÔE\ñô ð ð ð ×BÒBØ'Ø+Ô6Ø+Ô3Ø2Ô8ÔBØ2Ô8Ô=Ø2Ô=ÔBØ0ñô ð ùðO Ô&ñ /	ð /	ð /	ð /	ð /	r/   c           	     ót  — | j                              ¦   «         D �]\  }}|| j        v rŒ|                      ||¦  «         t	          || j                             ¦   «         ¦  «        }| j                             |¦  «         | j        |         j        }|j	        dk    r†t          |j        t          ¦  «        s,t          dt          |j        ¦  «        › d|j        ›�¦  «        ‚t!          |¦  «        }t"          j                             d|j        g|g||j        ¬¦  «        }nø|j	        dv r×|j        t"          j        j        t"          j        j        t"          j        j        hv rt5          d|j        › d�¦  «        ‚|j        |j        |j        g}t!          |¦  «        }|j        �1t"          j                             d	||g||j        | j        ¬
¦  «        }nCt"          j                             d	||g|| j        ¬¦  «        }nt5          d|j	        ›d�¦  «        ‚| j                             |¦  «         �ŒdS )zq
        Adds DQ ops (or Cast) for bias tensors that have been marked for quantization by op quantizers.
        ÚCastúUnexpected type z for input=)r½   Úto)NÚDequantizeLinearzUnexpected quantize type z for DequantizeLinear.Nr†  r6  )r7  zUnexpected operator type rª   ) rr   r  rŽ   Úquantize_bias_staticr   r�   rš   Úremove_initializerrJ   Ú	node_typer«   r@   r4  r¬   rœ   r4   r   r¾   r  r8  Úq_nameÚ
node_qtyper   r§   ÚBFLOAT16r¦   ÚRuntimeErrorr1  r2  r>   r|   Úadd_node)rA   rÍ   r  ÚinitÚquant_valuer5   rA  Úinputss           r0   r   z#QDQQuantizer._quantize_bias_tensors@  sa  € ð %)Ô$9×$?Ò$?Ñ$AÔ$Að 1	.ñ 1	.Ñ ˆI�yØ˜DÔ4Ð4Ð4Øà×%Ò% i°Ñ;Ô;Ð;Ý 	¨4¬:×+AÒ+AÑ+CÔ+CÑDÔDˆDØŒJ×)Ò)¨$Ñ/Ô/Ð/ØÔ2°9Ô=ÔFˆKØÔ$¨Ò.Ð.õ " $¤.µ#Ñ6Ô6ð rÝ#Ð$pµt¸D¼NÑ7KÔ7KÐ$pÐ$pÐXaÔXlÐ$pÐ$pÑqÔqÐqÝ.¨yÑ9Ô9�	Ý#œ{×4Ò4ØØ Ô'Ð(Ø�KØ"Ø”~ð  5ñ  ô  ��ð Ô&Ð*DÐDÐDØÔ)ÝÔ$Ô,ÝÔ$Ô-ÝÔ$Ô*ð.ð ð õ
 'Ð'qÀ;ÔCYÐ'qÐ'qÐ'qÑrÔrÐrØ%Ô,¨kÔ.DÀkÔFYÐZ�Ý.¨yÑ9Ô9�	ØÔ#Ð/Ý#'¤;×#8Ò#8Ø*ØØ"˜Ø!Ø(Ô-Ø#Ô1ð $9ñ $ô $�L�Lõ $(¤;×#8Ò#8Ø*ØØ"˜Ø!Ø#Ô1ð $9ñ $ô $�L�Lõ #Ð#Y¸{Ô?TÐ#YÐ#YÐ#YÑZÔZÐZØŒJ×Ò Ñ-Ô-Ð-Ñ-ðc1	.ð 1	.r/   c                ó&   — || j         v p|| j        v S r;   )rq   rr   r±   s     r0   Úis_tensor_quantizedz QDQQuantizer.is_tensor_quantizedw  s   € Ø˜dÔ6Ð6Ð^¸+ÈÔI^Ð:^Ð^r/   rÇ   r  ú
str | Noneútuple[bool, int | None]c                óê  — | j                              |¦  «        }|€dS | j                             |¦  «        rdS | j                             |¦  «        }| j        s|sdS |r| j                             ||¦  «        n|}|r(| j                             |¦  «        }|d         d         }t          |j	        ¦  «        }t          ||¦  «        \  }	}|	st          j        d|› d|› d|› �¦  «         dS d|fS )	aÞ  
        Checks if a given tensor is configured to be quantized per-channel. If so, also returns the channel axis.

        ORT only supports per-channel quantization on static weights (i.e., ONNX initializers). If the user did not provide
        tensor quantization overrides for this tensor, then the value of self.per_channel determines if the weight
        is to be quantized per-channel.

        Params:
            tensor_name: The name of the tensor to check.
            default_axis: The default channel axis. This method checks if the normalized axis is within bounds.
                          Can be overridden via the extra_options 'QDQOpTypePerChannelSupportToAxis'
                          and 'TensorQuantOverrides'.
            op_type: Optional, defaults to None. The operator type that is the only consumer of this weight.
                     Used to access the extra option 'QDQOpTypePerChannelSupportToAxis'.
        Returns:
            A tuple (is_per_channel, axis) in which the first element indicates whether the tensor is
            quantized per-channel and the second element is the channel axis.
            The returned axis is only None if the tensor is not per-channel or the axis is out of bounds.
        NrÚ   r   r>   zAxis z is out-of-range for weight 'z' with rank T)Úinitializersrt   rÉ   Úhas_per_tensor_overridesÚhas_per_channel_overridesr�   r{   Úget_per_channel_overridesrë   Údimsr    r‰   rŠ   )
rA   r    rÇ   r  Úweight_initializerÚhas_per_chan_overridesr>   Úper_chan_overridesÚweight_rankÚ
axis_valids
             r0   rË   z"QDQQuantizer.is_tensor_per_channelz  s,  € ð2 "Ô.×2Ò2°;Ñ?Ô?ÐØÐ%Ø�;àÔ&×?Ò?ÀÑLÔLð 	Ø�;à!%Ô!<×!VÒ!VÐWbÑ!cÔ!cÐØÔð 	Ð(>ð 	Ø�;àZaÐsˆtÔ;×?Ò?ÀÈÑVÔVÐVÐgsˆØ!ð 	1Ø!%Ô!<×!VÒ!VÐWbÑ!cÔ!cÐØ% aÔ(¨Ô0ˆDåÐ,Ô1Ñ2Ô2ˆÝ)¨$°Ñ<Ô<Ñˆ
�DØð 	ÝŒOÐm DÐmÐmÀ{ÐmÐmÐ`kÐmÐmÑnÔnÐnØ�;à�TˆzÐr/   rR   únp.ndarray | Nonec                óL  — | j                              ¦   «         }d}|| j        v r6| j        |                              |¦  «        j        }t          ||¦  «        }n8| j                             |d¦  «        }|rt          |j        d         |¦  «        }|�t          |¦  «        ndS )aø  
        Returns the quantization scale of a tensor that is consumed by the given node.
        :parameter tensor_name: The name of the tensor.
        :parameter consumer_node_name: The name of the node that consumes the tensor as input. Necessary in case
                                       the quantization type of the tensor was converted.
                                       Refer: QDQQuantizer::_add_qdq_ops_for_converted_activation.
        :returns: The quantization scale or None.
        Nr	   )
r�   rš   rŽ   rS   r1  r   rz   rt   r  r#   )rA   r    rR   r—  Úscale_initializerr1  Údq_nodes          r0   Ú_get_tensor_quantization_scalez+QDQQuantizer._get_tensor_quantization_scale«  s²   € ð ”z×-Ò-Ñ/Ô/ˆØ59Ðà˜$Ô2Ð2Ð2àÔ1°+Ô>×OÒOÐPbÑcÔcÔnˆJÝ ,¨Z¸Ñ FÔ FÐÐð Ô1×5Ò5°kÀ4ÑHÔHˆGØð QÝ$0°´¸qÔ1AÀ<Ñ$PÔ$PÐ!à;LÐ;XÕ$Ð%6Ñ7Ô7Ð7Ð^bÐbr/   rÍ   r  rD   c           
     ó  — || j         v r| j         |         j        j        S |                      |j        |j        ¦  «        }|€t          d|j        › d|› d�¦  «        ‚|                      |j        |j        ¦  «        }|€t          d|j        › d|› d�¦  «        ‚|                      ||||j	        ¦  «        \  }}}}}	}
t          ||||t          j        |j        dk    rdnd|	|
¬¦  «        }t          |dd¦  «        | j         |<   |S )	z]
        Quantized the bias. Zero Point == 0 and Scale == Input_Scale * Weight_Scale
        Nz9Unable to get valid quantization scale for weight input 'z' when quantizing bias 'z' to int32.z2Unable to get valid quantization scale for input 'r	   r   )r‰  r‹  )rŽ   rJ   rŠ  r¥  rE   r5   rd  r4   Úquantize_bias_static_implrG   r   r   rI  rÝ   r]   )rA   rÍ   r  rÕ   rÓ   Úquantized_bias_nameÚquantized_bias_scale_nameÚquantized_bias_zp_nameÚbias_scale_datar‰  r‹  rP  s               r0   r‡  z!QDQQuantizer.quantize_bias_staticÃ  s„  € ð ˜Ô0Ð0Ð0ØÔ+¨IÔ6Ô?ÔFÐFð ×:Ò:¸9Ô;PÐR[ÔReÑfÔfˆØÐÝð@ÈIÔLað @ð @Ø)2ð@ð @ð @ñô ð ð ×9Ò9¸)Ô:NÐPYÔPcÑdÔdˆØÐÝð@ÀYÔEYð @ð @Ø)2ð@ð @ð @ñô ð ð ×*Ò*¨9°kÀ<ÐQZÔQ_Ñ`Ô`ñ	
ØØ%Ø"ØØØõ )ØØØ%Ø"ÝÔ*Ø Ô%¨Ò)Ð)ˆAˆA¨tØØ!ð	
ñ 	
ô 	
ˆõ /FÀoÐW[Ð]aÑ.bÔ.bˆÔ  Ñ+à"Ð"r/   Ú Ú
param_namerJ  r   Úinit_name_suffixrW   c                ón  — |d         }|d         }|d         }|                      d¦  «        }|                      dd¦  «        }t          |¦  «        }	|	r|�t          |j        ¦  «        dk    sB|	s|�t          |j        ¦  «        d	k    s&|	s|€t          |j        ¦  «        dk    s
J d
¦   «         ‚t          |j        ¦  «        t          |j        ¦  «        k    s
J d¦   «         ‚|dz   |z   }
|dz   |z   }t          j                             |
||j        |                     ¦   «                              ¦   «         ¦  «        }| j	         
                    |¦  «         |j        t          j        k    rt          j        j        }nA|j        t          j        k    rt          j        j        }nt'          d|j        › d|›�¦  «        ‚t          j                             |||j        |                     ¦   «                              ¦   «         ¦  «        }| j	         
                    |¦  «         t)          ||¦  «        S )zù
        Creates and returns scale and zero-point initializers for the given quantization params. The initializers are
        named:
            - {param_name}_zero_point{init_name_suffix}
            - {param_name}_scale{init_name_suffix}
        rY   rX   r  r>   r   r   Nr'   r	   zWrong scale/zp shapesz,Scale and zero-point must have the same rankÚ_zero_pointÚ_scalezUnexpected dtype=z for param_name=)rt   r×   rë   rê   r¾   r  Úmake_tensorÚravelÚtolistr�   rÀ   rÜ   rÞ   Úfloat32r¥   r   r¦   Úfloat16r§   rd  rW   )rA   r­  rJ  r®  rY   rX   Úzero_point_typer>   r   Ú
is_blockedÚzero_point_namer1  Úinit_zprU  Ú
init_scales                  r0   rG  z(QDQQuantizer._make_scale_zp_initializersó  s.  € ð " ,Ô/ˆ
Ø˜WÔ%ˆØ& |Ô4ˆØ'×+Ò+¨FÑ3Ô3ˆØ&×*Ò*¨<¸Ñ;Ô;ˆ
Ý˜*Ñ%Ô%ˆ
àð	#Ø Ð,µ°U´[Ñ1AÔ1AÀQÒ1FÐ1FØð 2GØ#'Ð#3½¸E¼KÑ8HÔ8HÈAÒ8MÐ8MØð 9NØ#' <µC¸¼Ñ4DÔ4DÈÒ4IÐ4IÐ4IØ"ñ 5JÔ4IðKõ �5”;ÑÔ¥3 zÔ'7Ñ#8Ô#8Ò8Ð8Ð8Ð:hÑ8Ô8Ð8à$ }Ñ4Ð7GÑGˆØ (Ñ*Ð-=Ñ=ˆ
õ ”+×)Ò)Ø˜_¨jÔ.>À
×@PÒ@PÑ@RÔ@R×@YÒ@YÑ@[Ô@[ñ
ô 
ˆð 	Œ
×"Ò" 7Ñ+Ô+Ð+àŒ;�"œ*Ò$Ð$Ý#Ô/Ô5ˆJˆJØŒ[�BœJÒ&Ð&Ý#Ô/Ô7ˆJˆJåÐ\°´Ð\Ð\ÈjÐ\Ð\Ñ]Ô]Ð]Ý”[×,Ò,¨Z¸ÀUÄ[ÐRW×R]ÒR]ÑR_ÔR_×RfÒRfÑRhÔRhÑiÔiˆ
ØŒ
×"Ò" :Ñ.Ô.Ð.å% j°'Ñ:Ô:Ð:r/   ú#QDQTensorScaleZpInitializers | Nonec                óŒ  — | j         �	|| j         vrt          j        d|› d�¦  «         dS | j         |         }t          |t          ¦  «        s#t          dt          |¦  «        › d|›d�¦  «        ‚|                      ||j        ¦  «        }|j	        r|                      ||j	        d¦  «        nd}t          |||j        ¦  «        S )a  
        Create and returns all scale/zero_point initializers for a given tensor. If the tensor is converted
        to a different quantization type, this function creates two pairs of zp/scale initializers. Otherwise,
        only one pair of zp/scale initializers is created.
        Nz$Quantization parameters for tensor:"z" not specifiedr„  ú for rª   rb  )rŒ   r‰   rÊ   r«   rI   r¬   rœ   rG  rJ   rK   r[   rM   )rA   r    r�  Úoriginal_initsÚconverted_initss        r0   rz  z/QDQQuantizer._make_tensor_scale_zp_initializers  så   € ð Ô#Ð+¨{À$ÔBZÐ/ZÐ/ZÝŒLÐ\ÀÐ\Ð\Ð\Ñ]Ô]Ð]Ø�4àÔ0°Ô=ˆÝ˜-Õ)=Ñ>Ô>ð 	[ÝÐY­t°MÑ/BÔ/BÐYÐYÈÐYÐYÐYÑZÔZÐZà×9Ò9¸+À}ÔG]Ñ^Ô^ˆð Ô&ðˆD×,Ò,¨[¸-Ô:QÐS]Ñ^Ô^Ð^àð 	õ ,¨N¸OÈ]ÔMoÑpÔpÐpr/   Útensor_datar   Úquant_overridesúdict[str, Any]c                óö  — | j         }d|v r|d         j        }d|v rd|v r|d         |d         }}�n|t          j        j        k    rt          ||j        d         ¦  «        \  }}nÞ|                     d|j        d         ¦  «        }|                     d|j        d         ¦  «        }|                     d| j	        ¦  «        }|                     d	d
¦  «        }	t          ||	|¬¦  «        \  }
}t          |||
||| j        ¦  «        \  }}| j        r3|t          j        j        k    r|st          |||
|| j        ¬¦  «        \  }}t!          |                     ¦   «         |                     ¦   «         |¬¦  «        S )z”
        Calculates quantization parameters (scale/zero-point) given a tensor's min/max range and optional
        user-provided overrides.
        r  rX   rY   r	   ró   r   rô   Ú	symmetricr‘   F)r‘   rÅ  )ÚqminÚqmaxÚmin_real_range©rY   rX   r  )r‡   r<   r¾   r   ÚFLOAT8E4M3FNr   Úavg_stdrt   Úrange_valueÚis_activation_symmetricr   r   rÈ  Ú#is_activation_restricted_asymmetricÚUINT8r"   r   Úsqueeze)rA   rÁ  rÂ  r  ÚzerorX   ró   rô   rÅ  r‘   rÆ  rÇ  s               r0   Úcalc_quant_paramszQDQQuantizer.calc_quant_params4  sŠ  € ð
 Ô*ˆ
Ø˜?Ð*Ð*Ø(¨Ô6ÔBˆJà�oÐ%Ð%¨,¸/Ð*IÐ*IØ)¨,Ô7¸ÈÔ9Q�%ˆD‰DØ�4Ô+Ô8Ò8Ð8Ý1°*¸kÔ>QÐRSÔ>TÑUÔU‰KˆD�%�%à"×&Ò& v¨{Ô/FÀqÔ/IÑJÔJˆDØ"×&Ò& v¨{Ô/FÀqÔ/IÑJÔJˆDØ'×+Ò+¨K¸Ô9UÑVÔVˆIØ*×.Ò.¨~¸uÑEÔEˆLÝ0°È,ÐbkÐlÑlÔl‰JˆD�$Ý*¨4°°t¸TÀ9ÈdÔNaÑbÔb‰KˆD�%ØÔ7ð ¸JÍ$ÔJZÔJ`Ò<`Ð<`ÐirÐ<`å6Ø˜$ T°ÀTÔEXðñ ô ‘��eõ "¨T¯\ª\©^¬^À5Ç=Â=Á?Ä?Ð_iÐjÑjÔjÐjr/   údict[str, QDQTensorQuantParams]c                óì  — | j         €i S |                      ¦   «          i }| j         D ]Ì}| j         |         }t          |t          ¦  «        s#t	          dt          |¦  «        › d|›d�¦  «        ‚| j                             |i ¬¦  «        }|                      ||¦  «        }d}d}d|v r7|                      ||d         ¦  «        }|d          	                    d¦  «        }t          |||¦  «        ||<   ŒÍ|S )z´
        Calculates quantization parameters (scale/zero-point) for all tensors in the graph using each tensor's min/max range
        and optional user-provided overrides.
        Nr„  r¾  rª   )Údefault_valÚconvertÚ
recv_nodes)r’   Úadjust_tensor_rangesr«   r   r¬   rœ   rÉ   Úget_per_tensor_overridesrÒ  rt   rI   )rA   rŒ   r    ÚtdrÂ  rJ   rK   rM   s           r0   r‹   z$QDQQuantizer.calc_graph_quant_paramsP  s"  € ð
 ÔÐ%ØˆIà×!Ò!Ñ#Ô#Ð#à ÐØÔ-ð 	oð 	oˆKØÔ# KÔ0ˆBÝ˜b¥*Ñ-Ô-ð TÝÐ Rµ4¸±8´8Ð RÐ RÀ+Ð RÐ RÐ RÑSÔSÐSà"Ô9×RÒRÐS^ÐlnÐRÑoÔoˆOØ×-Ò-¨b°/ÑBÔBˆHØˆIØ#'Ð à˜OÐ+Ð+Ø ×2Ò2°2°ÀyÔ7QÑRÔR�	Ø'6°yÔ'A×'EÒ'EÀlÑ'SÔ'SÐ$å/CÀHÈiÐYmÑ/nÔ/nÐ Ñ,Ð,à"Ð"r/   údict[str, QuantizationParams]c                ót
  — i }| j                              ¦   «         D �]\  }}t          || j                             ¦   «         ¦  «        }|sŒ0t          |¦  «        }t          |j        ¦  «        }|j        t          j
        u }|r| j        n| j        }| j                             |¦  «        �r„| j        |         }	d|	d         v r|	d         d         j        }t          |         }
d|	d         v }|sZt!          t#          j        |	d         d         |
¬¦  «        t#          j        |	d         d         |j        ¦  «        |¬¦  «        ||<   någ }g }|	D ]d}|                     t#          j        |d         |
¦  «        ¦  «         |                     t#          j        |d         |j        ¬¦  «        ¦  «         Œe|	d         d         }t+          ||¦  «        \  }}|st-          d|j        › d	|› d
|› �¦  «        ‚t!          t#          j        |¦  «        t#          j        |¦  «        ||¬¦  «        ||<   �Œ| j                             |i g¦  «        }	d|	d         v r|	d         d         j        }|	d                              d|j        ¦  «        }|du}|p|r|                      |¦  «        n| j        }|	d                              d|¦  «        }|	d                              d| j        ¦  «        }d}d}|�|nd}|rx| j        dk    rmt+          ||¦  «        \  }}|st-          d|j        › d|› d
|› �¦  «        ‚|}t=          |||| j        |¦  «        \  }}t!          ||||| j        ¬¦  «        ||<   �ŒT|szt?          |                      ¦   «         |||| j!        |	d                              d¦  «        |	d                              d¦  «        ¬¦  «        \  }}t!          ||||¬¦  «        ||<   �ŒÐt+          ||¦  «        \  }}|st-          d|j        › d	|› d
|› �¦  «        ‚|}|j        |         }g }g }tE          |¦  «        D ]·}| #                    ||¦  «        }|	r|t          |	¦  «        k     r|	|         ni }t?          | $                    ¦   «         |||| j!        |                     d¦  «        |                     d¦  «        ¬¦  «        \  }}|                     |¦  «         |                     |¦  «         Œ¸t#          j%        |¦  «        }t#          j%        |¦  «        }t!          ||||¬¦  «        ||<   �Œ|S )ze
        Returns quantization parameters (scale/zero_point/quant_type) for all initializers.
        r  r   r>   rY   rÛ   rX   rÉ  zWeight z# has a per-channel axis with value z  that is out-of-bounds for rank )rY   rX   r  r>   NrÅ  r‘   z" has a block-wise axis with value )rY   rX   r  r>   r   ró   rô   )r‘   rÈ  Úrmin_overrideÚrmax_override)&rq   r  r   r�   rš   r#   rë   rê   r<   r&   r,   rˆ   r‡   rÉ   Úoverrides_scale_zpr   r   rÞ   rá   rÜ   r  r    rd  r½   rt   r>   Úis_weight_symmetricrÍ  r‘   r   r   r   ÚflattenrÈ  rì   Útaker³  r  )rA   rŒ   r    r{  rš   Úinitializer_dataÚinitializer_rankÚ	is_weightr  Ú	overridesÚzp_dtyperÎ   Úzero_points_listÚscales_listÚchan_overridesÚchannel_axisÚis_axis_validÚnorm_channel_axisÚis_symmetric_defaultÚis_symmetricr‘   rY   rX   Ú
block_axisÚnorm_block_axisÚchannel_countrý   Úper_channel_dataÚchannel_overridesÚchannel_zero_pointÚchannel_scales                                  r0   r  z+QDQQuantizer._calc_initializer_quant_paramsm  s†  € ð
 >@ÐØ(,Ô(@×(FÒ(FÑ(HÔ(Hð P	ñ P	Ñ$ˆK˜Ý& {°D´J×4JÒ4JÑ4LÔ4LÑMÔMˆKØð Øå4°[ÑAÔAÐÝ"Ð#3Ô#9Ñ:Ô:Ðð $Ô/Õ3EÔ3LÐLˆIØ.7ÐR˜Ô*Ð*¸TÔ=RˆJð Ô*×=Ò=¸kÑJÔJñ #Ø Ô7¸ÔD�	Ø 9¨Q¤<Ð/Ð/Ø!*¨1¤¨lÔ!;Ô!G�Jå/°
Ô;�Ø!'¨9°Q¬<Ð!7�Ø%ð Ý7IÝ#%¤8¨I°a¬L¸Ô,FÈhÐ#WÑ#WÔ#WÝ œh y°¤|°GÔ'<Ð>NÔ>TÑUÔUØ#-ð8ñ 8ô 8Ð'¨Ñ4Ð4ð (*Ð$Ø"$�KØ*3ð lð l˜Ø(×/Ò/µ´¸ÈÔ9UÐW_Ñ0`Ô0`ÑaÔaÐaØ#×*Ò*­2¬8°NÀ7Ô4KÐScÔSiÐ+jÑ+jÔ+jÑkÔkÐkÐkà#,¨Q¤<°Ô#7�LÝ7EÀlÐTdÑ7eÔ7eÑ4�MÐ#4Ø(ð Ý(ðI kÔ&6ð Ið IÐ[gð Ið IØ6FðIð Iñô ð õ
 8JÝ#%¤8Ð,<Ñ#=Ô#=Ý œh {Ñ3Ô3Ø#-Ø.ð	8ñ 8ô 8Ð'¨Ñ4ñ ð Ô3×7Ò7¸ÀbÀTÑJÔJˆIØ˜y¨œ|Ð+Ð+Ø& qœ\¨,Ô7ÔC�
à$ Qœ<×+Ò+¨F°KÔ4DÑEÔEˆLØ)°Ð5ˆNð $2ð $Ø8AÐc�×(Ò(¨Ñ4Ô4Ð4ÀtÔGcð !ð % Qœ<×+Ò+¨KÐ9MÑNÔNˆLØ$ Qœ<×+Ò+¨N¸DÔ<MÑNÔNˆLØ,0ˆJØ'+ˆEð *6Ð)A˜˜ÀqˆJØð I˜Tœ_¨qÒ0Ð0å1?À
ÐL\Ñ1]Ô1]Ñ.�˜Ø$ð Ý$ðE +Ô"2ð Eð EÐV`ð Eð EØ2BðEð Eñô ð ð  /�Ý$<Ø$ØØ Ø”OØ ñ%ô %Ñ!�
˜Eõ 4FØ)ØØ)Ø%Ø#œð4ñ 4ô 4Ð# KÑ0Ñ0ð $ð 2Ý$=Ø$×,Ò,Ñ.Ô.ØØ Ø!-Ø#'Ô#6Ø"+¨A¤,×"2Ò"2°6Ñ":Ô":Ø"+¨A¤,×"2Ò"2°6Ñ":Ô":ð%ñ %ô %Ñ!�
˜Eõ 4FØ)ØØ)Ø%ð	4ñ 4ô 4Ð# KÑ0Ñ0õ 4BÀ,ÐP`Ñ3aÔ3aÑ0�Ð0Ø$ð Ý$ðE +Ô"2ð Eð EÐWcð Eð EØ2BðEð Eñô ð ð
  1�Ø 0Ô 6°|Ô D�Ø#%Ð Ø �Ý˜}Ñ-Ô-ð 6ð 6�AØ'7×'<Ò'<¸QÀÑ'MÔ'MÐ$Ø8AÐ(`ÀaÍ#ÈiÉ.Ì.ÒFXÐFX¨	°!¬¨Ð^`Ð%Ý8QØ(×.Ò.Ñ0Ô0Ø"Ø$Ø%1Ø'+Ô':Ø&7×&;Ò&;¸FÑ&CÔ&CØ&7×&;Ò&;¸FÑ&CÔ&Cð9ñ 9ô 9Ñ5Ð&¨ð %×+Ò+Ð,>Ñ?Ô?Ð?Ø×&Ò& }Ñ5Ô5Ð5Ð5åœZÐ(8Ñ9Ô9�
Ýœ
 ;Ñ/Ô/�Ý3EØ)ØØ)Ø%ð	4ñ 4ô 4Ð# KÑ0Ñ0ð #Ð"r/   r;   )r    r3   )r³   r3   r4   r3   r5   r3   )rš   rº   rN   rº   )rÅ   )rÓ   rÔ   rÕ   rÔ   rE   r3   rÖ   rº   rÎ   r×   rN   rØ   )Nr   )r.  r3   r/  r3   r0  r3   r1  r3   r2  r3   r>   r3  r   r4  )r=  r3   r>  r3   r?  r3   r1  r3   r2  r3   r>   r3  r   r4  )r   r4  )rE  rº   )r    r3   rÇ   r4  r  r”  rN   r•  )r    r3   rR   r3   rN   r¡  )rÍ   r3   r  rD   rN   r3   )r¬  )r­  r3   rJ  r   r®  r3   rN   rW   )r    r3   rN   r¼  )rÁ  r   rÂ  rÃ  rN   r   )rN   rÓ  )rN   rÛ  )'r(   r)   r*   rB   r£   r¨   r&   r+   r°   r²   rµ   r·   r¹   rÄ   rÒ   rÿ   r  r  r  r&  r-  r<  rB  rD  rQ  r^  rw  r  r  r   r“  rË   r¥  r‡  rG  rz  rÒ  r‹   r  r.   r/   r0   r`   r`   ‘   s&  € € € € € ð ðc&ð c&ð c&ð c&ðJð ð ðð ð ð, EIÐVhÔVsð yð yð yð yð:Xð Xð Xð Xð
ð 
ð 
ð 
ð"Tð Tð Tð Tðyð yð yð
ð 
ð 
ð 
ð/mð /mð /mð /mðbN-ð N-ð N-ð N-ð`1@ð 1@ð 1@ðf*ð *ð *ð6ð 6ð 6ð ð  ð  ð>ð ð ð,  Øð-ð -ð -ð -ð -ð@  Øð-ð -ð -ð -ð -ðF Øð;ð ;ð ;ð ;ð ;ð@Deð Deð Deð DeðLAið Aið Aið AiðF[
ð [
ð [
ðz4:ð 4:ð 4:ðl6ð 6ð 6ðp5.ð 5.ð 5.ðn_ð _ð _ð _ð #ð	/ð /ð /ð /ð /ðbcð cð cð cð0.#ð .#ð .#ð .#ðb Z\ð(;ð (;ð (;ð (;ð (;ðTqð qð qð qð.kð kð kð kð8#ð #ð #ð #ð:X#ð X#ð X#ð X#ð X#ð X#r/   r`   )7Ú
__future__r   r‰   Údataclassesr   Úenumr   Útypingr   ÚnumpyrÞ   r¾   r   r   r¥   Úbase_quantizerr
   r   Ú	calibrater   Úquant_utilsr   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r    r!   r"   r#   Úregistryr$   r&   r2   r9   rD   rI   rW   r[   r]   r`   r.   r/   r0   ú<module>r      s¡  ðð #Ð "Ð "Ð "Ð "Ð "à €€€Ø !Ð !Ð !Ð !Ð !Ð !Ø Ð Ð Ð Ð Ð Ø Ð Ð Ð Ð Ð à Ð Ð Ð Ø €€€Ø Ð Ð Ð Ð Ð Ø &Ð &Ð &Ð &Ð &Ð &à =Ð =Ð =Ð =Ð =Ð =Ð =Ð =Ø !Ð !Ð !Ð !Ð !Ð !ðð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð2 )Ð (Ð (Ð (Ð (Ð (ðð ð ð ð ˜ñ ô ð ð ðð ð ð ð ñ ô ñ „ðð#ð #ð #ð #ð #ñ #ô #ð #ð ðð ð ð ð ñ ô ñ „ðð ðfð fð fð fð fñ fô fñ „ðfð& ðð ð ð ð ñ ô ñ „ðð ð*ð *ð *ð *ð *ñ *ô *ñ „ð*ð ðfð fð fð fð fñ fô fñ „ðfð$t#ð t#ð t#ð t#ð t#�=ñ t#ô t#ð t#ð t#ð t#r/   