Ë
    –\;jÿu  ã                   óº   — d dl Z d dlZd dlmZ d dlmZ d dlmZ ddlm	Z	m
Z
 ddlmZmZmZmZ dd	lmZmZ d
ZdZd„ Z G d„ d«      Z G d„ d«      Z G d„ d«      Zy)é    N)ÚIrGraph)Úcore)Úquant_layersé   )ÚQuantWeightPassÚReplaceFakeQuantDequantPass)Ú_get_input_name_indexÚ_get_op_input_var_namesÚ_get_output_name_indexÚ$move_persistable_var_to_global_blocké   )Ú
fuse_utilsÚutilsú.pdmodelz
.pdiparamsc                 ó^  — ddl m} |j                  j                  j                  j
                  | d<   |j                  j                  j                  j                  | d<   |j                  |j                  j                  «       |j                  |j                  j
                  «       | |fS )Nr   )ÚfleetÚColumnParallelLinearÚRowParallelLinear)Úpaddle.distributedr   Úmeta_parallelÚparallel_layersÚ	mp_layersr   r   Úappend)Úlayer_name_mapÚfake_quant_input_layersr   s      úkG:\00. PROJECTS\API\Inventory\templateJSON\kerjaOCR\Lib\site-packages\paddle/quantization/imperative/qat.pyÚlazy_import_fleetr   &   s›   € Ý(ð 	×Ñ×+Ñ+×5Ñ5×JÑJð Øñð
 	×Ñ×+Ñ+×5Ñ5×GÑGð Øñð ×"Ñ" 5×#6Ñ#6×#HÑ#HÔIØ×"Ñ" 5×#6Ñ#6×#KÑ#KÔLØÐ2Ð2Ð2ó    c                   óN   ‡ — e Zd ZdZg d¢dddddddddddfˆ fd	„	Zd
„ Zdd„Zˆ xZS )ÚImperativeQuantAwarezI
    Applying quantization aware training (QAT) to the dgraph model.
    )ÚConv2DÚLinearÚConv2DTransposer   r   Úabs_maxÚmoving_average_abs_maxé   çÍÌÌÌÌÌì?FNc                 óŽ   •— t         ‰| �  «        || _        ||||||||	|
|dœ
}t        di |¤Ž| _        t        |||«      | _        y)aŒ  
        The constructor for ImperativeQuantAware.

        Args:
            quantizable_layer_type(list[str | layer]): List the type of
                layers that will be quantized. Default is ['Conv2D', 'Linear'].
            weight_quantize_type(str): quantization type for weights,
                which supports 'abs_max' and 'channel_wise_abs_max'.
            activation_quantize_type(str): quantization type for activations,
                which supports 'abs_max' and 'moving_average_abs_max' now.
                If using 'abs_max' mode, the quantization scale will be
                calculated dynamically each step in both training and testing
                period. If using 'moving_average_abs_max', the static
                quantization scale will be calculated during training and
                used in inference.
            weight_bits(int): quantization bit number for weights, whereas
                the bias is not quantized.
            activation_bits(int): quantization bit number for activations.
            moving_rate(float): the parameter for 'moving_average_abs_max'
                quantization.
            fuse_conv_bn(bool): Whether to fuse conv and bn, default is False.
            weight_preprocess_layer(paddle.nn.Layer, optional): A paddle
                Layer that defines how to preprocess weight before quantization.
                Using this can quickly test if user's preprocess method works
                or not. The input is non-quantized weight and function returns
                processed weight to be quantized.
                If None, the weight will be quantized directly.
                Default is None.
            act_preprocess_layer(paddle.nn.Layer, optional): A paddle Layer
                that defines how to preprocess activation before quantization.
                Using this can quickly test if user's preprocess method works
                or not. The input is non-quantized activation and function returns
                processed activation to be quantized.
                If None, the activation will be quantized directly.
                Default is None.
            weight_quantize_layer(paddle.nn.Layer, optional): A paddle Layer that
                defines how to quantize weight.
                Using this can quickly test if user's quantization method works or not.
                In this layer, user should both define quantization method and
                dequantization method, that is, the function's input is non-quantized
                weight and returns dequantized weight.
                If None, will use uantization op defined by 'weight_quantize_type'.
                Default is None.
            act_quantize_layer(paddle.nn.Layer, optional): A paddle Layer that defines
                how to quantize activation.
                Using this can quickly test if user's quantization method works or not.
                In this layer, user should both define quantization method and
                dequantization method, that is, the function's input is non-quantized
                activation and returns dequantized activation.
                If None, will use quantization op defined by 'activation_quantize_type'.
                Default is None.
            onnx_format (bool, optional): Whether to export the quantized model
                with format of ONNX. Default is False.

        Note:
            If user sets attribute 'skip_quant' to a Layer that support dynamic
            quantization and sets it to true, the layer would not be quantized
            during training. If this attribute is not sets or the attribute is
            false, the Layer would be qunatized in training.

        Examples:
            .. code-block:: python

                >>> import paddle
                >>> from paddle.static.quantization import (
                ...     ImperativeQuantAware,
                ... )
                >>> from paddle.vision.models import (
                ...     resnet,
                ... )

                >>> model = resnet.resnet50(pretrained=True)

                >>> imperative_qat = ImperativeQuantAware(
                ...     weight_quantize_type='abs_max',
                ...     activation_quantize_type='moving_average_abs_max')

                >>> # Add the fake quant logical.
                >>> # The original model will be rewrite.
                >>> # The outscale of outputs in supportted layers would be calculated.
                >>> imperative_qat.quantize(model)

                >>> # Fine-tune the quantized model
                >>> # ...

                >>> # Save quant model for the inference.
                >>> imperative_qat.save_quantized_model(
                ...     layer=model,
                ...     model_path="./resnet50_qat",
                ...     input_spec=[
                ...         paddle.static.InputSpec(
                ...         shape=[None, 3, 224, 224], dtype='float32')])

            .. code-block:: python

                >>> import paddle
                >>> from paddle.static.quantization import (
                ...     ImperativeQuantAware,
                ... )

                >>> class ImperativeModel(paddle.nn.Layer):
                ...     def __init__(self):
                ...         super().__init__()
                ...         # self.linear_0 would skip the quantization.
                ...         self.linear_0 = paddle.nn.Linear(784, 400)
                ...         self.linear_0.skip_quant = True

                ...         # self.linear_1 would not skip the quantization.
                ...         self.linear_1 = paddle.nn.Linear(400, 10)
                ...         self.linear_1.skip_quant = False

                ...     def forward(self, inputs):
                ...         x = self.linear_0(inputs)
                ...         x = self.linear_1(inputs)
                ...         return x

                >>> model = ImperativeModel()
                >>> imperative_qat = ImperativeQuantAware(
                ...     weight_quantize_type='abs_max',
                ...     activation_quantize_type='moving_average_abs_max')

                >>> # Add the fake quant logical.
                >>> # The original model will be rewrite.
                >>> #
                >>> # There is only one Layer(self.linear1) would be added the
                >>> # fake quant logical.
                >>> imperative_qat.quantize(model)

                >>> # Fine-tune the quantized model
                >>> # ...

                >>> # Save quant model for the inference.
                >>> imperative_qat.save_quantized_model(
                ...    layer=model,
                ...    model_path="./imperative_model_qat")
        )
Úquantizable_layer_typeÚweight_quantize_typeÚactivation_quantize_typeÚweight_bitsÚactivation_bitsÚmoving_rateÚweight_preprocess_layerÚact_preprocess_layerÚweight_quantize_layerÚact_quantize_layerN© )ÚsuperÚ__init__Úfuse_conv_bnÚImperativeQuantizeInputsÚ_quantize_inputsÚImperativeQuantizeOutputsÚ_quantize_outputs)Úselfr)   r*   r+   r,   r-   r.   r6   r/   r0   r1   r2   Úonnx_formatÚkwargsÚ	__class__s                 €r   r5   zImperativeQuantAware.__init__9   sf   ø€ ôz 	‰ÑÔØ(ˆÔð '=Ø$8Ø(@Ø&Ø.Ø&Ø'>Ø$8Ø%:Ø"4ñ
ˆô !9Ñ B¸6Ñ BˆÔä!:Ø˜¨+ó"
ˆÕr   c                 ó
  — t        |t        j                  j                  «      sJ d«       ‚| j                  rt        j                  |«       | j                  j                  |«       | j                  j                  |«       |S )a¤  
        According to weights' and activations' quantization types,
        the model will be added some fake quant ops, such as
        fake_quantize_dequantize_moving_average_abs_max,
        fake_quantize_dequantize_abs_max and so on. At the same time,
        the out_scale value of outputs would be calculated.

        Args:
            model(paddle.nn.Layer): the model to be quantized.
        Returns:
            None

        Examples:
            .. code-block:: python

                >>> import paddle
                >>> from paddle.static.quantization import (
                ...     ImperativeQuantAware,
                ... )

                >>> class ImperativeModel(paddle.nn.Layer):
                ...     def __init__(self):
                ...         super().__init__()
                ...         # self.linear_0 would skip the quantization.
                ...         self.linear_0 = paddle.nn.Linear(784, 400)
                ...         self.linear_0.skip_quant = True

                ...         # self.linear_1 would not skip the quantization.
                ...         self.linear_1 = paddle.nn.Linear(400, 10)
                ...         self.linear_1.skip_quant = False

                ...     def forward(self, inputs):
                ...         x = self.linear_0(inputs)
                ...         x = self.linear_1(inputs)
                ...         return x

                >>> model = ImperativeModel()
                >>> imperative_qat = ImperativeQuantAware(
                ...     weight_quantize_type='abs_max',
                ...     activation_quantize_type='moving_average_abs_max')

                >>> # Add the fake quant logical.
                >>> # The original model will be rewrite.
                >>> #
                >>> # There is only one Layer(self.linear1) would be added the
                >>> # fake quant logical.
                >>> imperative_qat.quantize(model)
        ú2The model must be the instance of paddle.nn.Layer.)	Ú
isinstanceÚpaddleÚnnÚLayerr6   r   r8   Úapplyr:   )r;   Úmodels     r   ÚquantizezImperativeQuantAware.quantizeì   st   € ôb Ø”6—9‘9—?‘?ô
ð 	@à?ó	@ð 
ð ×ÒÜ×#Ñ# EÔ*à×Ñ×#Ñ# EÔ*Ø×Ñ×$Ñ$ UÔ+Øˆr   c                 óB   —  | j                   j                  |||fi |¤Ž y ©N)r:   Úsave_quantized_model)r;   ÚlayerÚpathÚ
input_specÚconfigs        r   rJ   z)ImperativeQuantAware.save_quantized_model(  s'   € Ø3ˆ×Ñ×3Ñ3Ø�4˜ñ	
Ø'-ó	
r   rI   )Ú__name__Ú
__module__Ú__qualname__Ú__doc__r5   rG   rJ   Ú__classcell__©r>   s   @r   r    r    4   sB   ø„ ñò 
ð 'Ø!9ØØØØØ $Ø!Ø"ØØõ'q
òf:÷x
r   r    c            
       óH   ‡ — e Zd ZdZg d¢dddddddddf
ˆ fd„	Zd	„ Zd
„ Zˆ xZS )r7   z€
    Based on the input params, add the quant_dequant computational
    logic both for activation inputs and weight inputs.
    )r!   r"   r#   r$   r%   r&   r'   Nc           
      óf  •‡ — t         ‰‰ �  «        t        t        j                  t        j
                  «      \  ‰ _        ‰ _        t        ˆ fd„|D «       «      ‰ _        ‰ j                  D ]*  }t        |t        «      s|‰ j
                  v rŒ"J d|z  «       ‚ h d£}ddh}|dk7  r||v s
J d|z  «       ‚||v s
J d|z  «       ‚d„ } ||«      sJ d	«       ‚ ||«      sJ d
«       ‚d„ } ||«      sJ d«       ‚ ||«      sJ d«       ‚ ||	«      sJ d«       ‚ ||
«      sJ d«       ‚||||||||	|
dœ	‰ _
        y)zz
        The constructor for ImperativeQuantizeInputs.

        Please refer to the args of ImperativeQuantAware.
        c              3   ó\   •K  — | ]#  }|‰j                   v r‰j                   |   n|–— Œ% y ­wrI   )r   )Ú.0rK   r;   s     €r   Ú	<genexpr>z4ImperativeQuantizeInputs.__init__.<locals>.<genexpr>K  s@   øè ø€ ð -
ñ 0�ð ˜×+Ñ+Ñ+ð ×Ñ Ò&àóñ 0ùs   ƒ),z!%s is unspported to be quantized.>   r$   Ú
lsq_weightÚchannel_wise_abs_maxr%   Úchannel_wise_lsq_weightr%   Úlsq_actzUUnsupported weight_quantize_type: %s. It can only be abs_max or channel_wise_abs_max.z_Unsupported activation_quantize_type: %s. It can only be moving_average_abs_max or lsq_act now.c                 ó>   — t        | t        «      xr | dk\  xr | dk  S )Nr   é   )rA   Úint)Úbitss    r   Ú<lambda>z3ImperativeQuantizeInputs.__init__.<locals>.<lambda>n  s    € œ D¬#Ó.ÒK°4¸1±9ÒKÀÈÁÐKr   z%weight_bits should be 1, 2,... or 16.z)activation_bits should be 1, 2,... or 16.c                 óV   — | d u xs$ t        | t        j                  j                  «      S rI   )Ú
issubclassrB   rC   rD   )Úmethods    r   rb   z3ImperativeQuantizeInputs.__init__.<locals>.<lambda>u  s&   €  V¨t ^ò &
´zØ”F—I‘I—O‘Oó8
ð &
r   z%weight_preprocess should be nn.Layer.z"act_preprocess should be nn.Layer.z#weight_quantize should be nn.Layer.z act_quantize should be nn.Layer.)	r*   r+   r,   r-   r.   Úweight_pre_layerÚact_pre_layerÚweight_quant_layerÚact_quant_layerN)r4   r5   r   r   r   r   ÚtupleÚ_quantizable_layer_typerA   ÚstrÚ_kwargs)r;   r)   r*   r+   r,   r-   r.   r/   r0   r1   r2   rK   Úquantize_typeÚact_quantize_typeÚ
bits_checkÚlayer_checkr>   s   `               €r   r5   z!ImperativeQuantizeInputs.__init__4  så  ù€ ô$ 	‰ÑÔÜ<MÜ× Ñ ¤%×"?Ñ"?ó=
Ñ9ˆÔ˜TÔ9ô (-ó -
ñ 0ó	-
ó (
ˆÔ$ð ×1Ô1ˆEä˜u¤cÔ*Ø˜T×9Ñ9Ò9ð=ð 4°eÑ;ó=ð:ð 2ò
ˆð 6°yÐAÐà Ð$<Ò<Ø$¨Ñ5ð	
ð2Ø4HñIó		
ð6ð (Ð+<Ñ<ð 	
ð=à&ñ'ó	
Ð<ñ Lð 	ñ ˜+Ô&ÐOÐ(OÓOÐ&ÙØô
ð 	7à6ó	7ð 
ñ
ˆñ Ø#ô
ð 	3à2ó	3ð 
ñ Ø ô
ð 	0à/ó	0ð 
ñ Ø!ô
ð 	1à0ó	1ð 
ñ Øô
ð 	.à-ó	.ð 
ð
 %9Ø(@Ø&Ø.Ø&Ø 7Ø1Ø"7Ø1ñ

ˆ�r   c                 óZ  — t        |t        j                  j                  «      sJ d«       ‚|j	                  «       D ]m  \  }}t        || j
                  «      rt        |d«      r|j                  du rŒ7t        j                  ||«      \  }}| j                  |«      }t        |||«       Œo y)a  
        Quantize the weights and activations to calculate for specific
        layers.

        Args:
            model(paddle.nn.Layer): The target model which would
                calculate the input quantization scale.

        Returns:
            None
        r@   Ú
skip_quantTN)rA   rB   rC   rD   Únamed_sublayersrk   Úhasattrrs   r   Úfind_parent_layer_and_sub_nameÚ_get_input_quantized_layerÚsetattr)r;   rF   ÚnameÚ	cur_layerÚparent_layerÚsub_nameÚcur_quant_layers          r   rE   zImperativeQuantizeInputs.apply‘  s©   € ô Ø”6—9‘9—?‘?ô
ð 	@à?ó	@ð 
ð  %×4Ñ4Ö6‰OˆD�)Ü˜i¨×)EÑ)EÔFÜ˜	 <Ô0Ø×(Ñ(¨DÑ0àä%*×%IÑ%IØ�tó&Ñ"ˆL˜(ð #×=Ñ=¸iÓHˆOÜ�L (¨OÕ<ñ  7r   c                 óê   — d }| j                   j                  «       D ]  \  }}t        ||«      sŒd|z   } n |€J d|j                  «       z  «       ‚t	        j
                  |   |fi | j                  ¤ŽS )NÚ	Quantizedz,The layer %s is unsupported to be quantized.)r   ÚitemsrA   Ú	full_namer   Ú__dict__rm   )r;   rK   Úquant_layer_nameÚkeyÚvalues        r   rw   z3ImperativeQuantizeInputs._get_input_quantized_layer°  sƒ   € ØÐà×-Ñ-×3Ñ3Ö5‰JˆC�Ü˜% Õ'Ø#.°Ñ#4Ð Ùð 6ð  Ð+ð 	
Ø:¸U¿_¹_Ó=NÑNó	
Ð+ô ×$Ñ$Ð%5Ñ6°uÑMÀÇÁÑMÐMr   )rO   rP   rQ   rR   r5   rE   rw   rS   rT   s   @r   r7   r7   .  s;   ø„ ñò  GØ&Ø!9ØØØØ $Ø!Ø"Øõ[
òz=ö>Nr   r7   c                   óJ   ‡ — e Zd ZdZd	ˆ fd„	Zd„ Zd
d„Zd„ Zd„ Zd„ Z	d„ Z
ˆ xZS )r9   z8
    Calculate the output scales for target layers.
    c                 óL   •— t         ‰| �  «        || _        || _        || _        y)a4  
        The constructor for ImperativeQuantizeOutputs.

        Args:
            moving_rate(float): The decay coefficient of moving average.
                                The default value is 0.9.
            activation_bits(int, optional): quantization bit number for activation. Default is 8.
        N)r4   r5   Ú_moving_rateÚ_activation_bitsÚ_onnx_format)r;   r.   r-   r<   r>   s       €r   r5   z"ImperativeQuantizeOutputs.__init__Ã  s(   ø€ ô 	‰ÑÔØ'ˆÔØ /ˆÔØ'ˆÕr   c                 óØ  — t        |t        j                  j                  «      sJ d«       ‚|j	                  «       D ]¬  \  }}d|v rŒ| j                  |«      sŒt        j                  ||«      \  }}d}t        |t        t        j                  «      «      r#t        j                  || j                  |¬«      }n"t        j                  || j                  |¬«      }t        |||«       Œ® y)aB  
        Insert the `moving_average_abs_max_scale` layers to calculate the
        output scales for specific layers in the dygraph model.

        Args:
            model(paddle.nn.Layer): The target model which would be
                calculate the output quantization scale.

        Returns:
            None
        r@   Ú_act_preprocessN)Úreduce_type)rA   rB   rC   rD   rt   Ú_is_target_layerr   rv   rj   Úfake_quant_output_layersr   ÚFakeQuantMAOutputScaleLayerrˆ   ÚMAOutputScaleLayerrx   )r;   rF   Úcur_namerz   r{   r|   r�   r}   s           r   rE   zImperativeQuantizeOutputs.applyÑ  sã   € ô Ø”6—9‘9—?‘?ô
ð 	@à?ó	@ð 
ð $)×#8Ñ#8Ö#:ÑˆH�iØ  HÑ,ØØ×(Ñ(¨Ô3Øä%*×%IÑ%IØ�xó&Ñ"ˆL˜(ð ˆKä˜)¤U¬5×+IÑ+IÓ%JÔKÜ".×"JÑ"JØ˜t×0Ñ0¸kô#‘ô #/×"AÑ"AØ˜t×0Ñ0¸kô#�ô �L (¨OÕ<ñ+ $;r   c                 ó†  — t        |t        j                  j                  «      sJ d«       ‚|r!t        j                  j                  ||¬«       t        j                  j                  d|||dœ|¤Ž d}t        j                  «       rd}t        j                  «        t        j                  «       }t        j                  j                  «       }t        j                  j                  |«      }t        j                  j!                  |«      }	t        j                  j#                  |«      }
|
t$        z   }|
t&        z   }t        j                  j)                  |	|||¬«      \  }}}| j*                  sÀ| j-                  |||«       t/        t        j0                  |j2                  «      d¬«      }|j5                  «       D ]L  }|j7                  «       D ]'  }|j9                  «       dk(  sŒ|j;                  |«       Œ) |j=                  «        ŒN |j?                  «       }| jA                  |«       d}nºt/        t        j0                  |j2                  «      d¬«      }tC        ||| jD                  ¬	«      }|j5                  «       D ]  }d|_#        |jI                  |«       Œ tK        ||«      }|j5                  «       D ]  }d|_#        |jI                  |«       Œ |j?                  «       }d}tM        |«       d
}|€d}n)|jO                  d«      r|jQ                  dd«      d   }n|}t        j                  jS                  |	|«      }|D �cg c]!  }|jU                  «       jW                  |«      ‘Œ# }}t        j                  jY                  |||||j[                  «       |¬«       |rt        j\                  «        y
y
c c}w )a�  
        Save the quantized model for the inference.

        Args:
            model (Layer): The model to be saved.
            path (str): The path prefix to save model. The format is
                ``dirname/file_prefix`` or ``file_prefix``.
            input_spec (list[InputSpec|Tensor], optional): Describes the input
                of the saved model's forward method, which can be described by
                InputSpec or example Tensor. If None, all input variables of
                the original Layer's forward method would be the inputs of
                the saved model. Default None.
            **config (dict, optional): Other save configuration options for
                compatibility. We do not recommend using these configurations,
                they may be removed in the future. If not necessary, DO NOT use
                them. Default None.
                The following options are currently supported:
                (1) output_spec (list[Tensor]): Selects the output targets of
                the saved model. By default, all return variables of original
                Layer's forward method are kept as the output of the saved model.
                If the provided ``output_spec`` list is not all output variables,
                the saved model will be pruned according to the given
                ``output_spec`` list.

        Returns:
            None
        r@   )rM   )rK   rL   rM   FT)ÚexecutorÚmodel_filenameÚparams_filename)Úfor_testÚmoving_average_abs_max_scale)Ú
quant_bitsNrF   r   Ú.r   r   )r”   ÚprogramÚ
clip_extrar3   )/rA   rB   rC   rD   ÚjitÚ	to_staticÚsaveÚin_dynamic_modeÚenable_staticr   ÚCPUPlaceÚstaticÚglobal_scopeÚExecutorÚosrL   ÚdirnameÚbasenameÚINFER_MODEL_SUFFIXÚINFER_PARAMS_SUFFIXÚload_inference_modelrŠ   Ú_gather_scalesr   ÚGraphÚdescÚall_sub_graphsÚall_op_nodesry   Úsafe_remove_nodesÚresolve_hazardÚ
to_programÚ_set_skip_quant_attrr   r‰   Ú	_for_testrE   r   r   ÚendswithÚrsplitÚjoinÚglobal_blockÚvarÚsave_inference_modelÚcloneÚdisable_static)r;   rF   rL   rM   rN   Úis_dynamic_modeÚplaceÚscopeÚexer§   r¨   r•   r–   Úinfer_programÚfeed_target_namesÚfetch_targetsÚgraphÚ	sub_graphÚ_oprœ   Útransform_passÚquant_weight_passÚ
model_nameÚpath_prefixry   Ú	feed_varss                             r   rJ   z.ImperativeQuantizeOutputs.save_quantized_modelø  sU  € ô8 Ø”6—9‘9—?‘?ô
ð 	@à?ó	@ð 
ñ Ü�J‰J× Ñ  °:Ð Ô>Ü�
‰
�‰ÐP˜e¨$¸:ÑPÈÒPàˆÜ×!Ñ!Ô#Ø"ˆOÜ× Ñ Ô"ä—‘“ˆÜ—‘×*Ñ*Ó,ˆÜ�m‰m×$Ñ$ UÓ+ˆä—'‘'—/‘/ $Ó'ˆÜ—7‘7×#Ñ# DÓ)ˆØ!Ô$6Ñ6ˆØ"Ô%8Ñ8ˆô �M‰M×.Ñ.ØØØ)Ø+ð	 /ó 
ñ		
ØØØð × Ò Ø×Ñ ¨u°mÔDô œDŸJ™J }×'9Ñ'9Ó:ÀUÔKˆEØ"×1Ñ1Ö3�	Ø$×1Ñ1Ö3�CØ—x‘x“zÐ%CÓCØ!×3Ñ3°CÕ8ð 4ð ×(Ñ(Õ*ð	 4ð
 "×,Ñ,Ó.ˆMà×%Ñ% mÔ4à‰JäœDŸJ™J }×'9Ñ'9Ó:ÀUÔKˆEÜ8Ø�u¨×)>Ñ)>ôˆNð #×1Ñ1Ö3�	Ø&*�	Ô#Ø×$Ñ$ YÕ/ð 4ô !0°°uÓ =ÐØ"×1Ñ1Ö3�	Ø&*�	Ô#Ø!×'Ñ'¨	Õ2ð 4ð "×,Ñ,Ó.ˆMàˆJä,¨]Ô;àˆ
ØÐ!Ø ‰JØ×$Ñ$ ZÔ0Ø'×.Ñ.¨s°AÓ6°qÑ9‰Jà'ˆJÜ—g‘g—l‘l 7¨JÓ7ˆá?Pó
Ù?P°tˆM×&Ñ&Ó(×,Ñ,¨TÕ2Ð?Pð 	ð 
ô 	�‰×*Ñ*ØØØØØ!×'Ñ'Ó)Ø!ð 	+ô 	
ñ Ü×!Ñ!Õ#ð ùò
s   Í&N>c                 óØ  — t        |t        j                  j                  «      sy| j                  r't        |t        t        j                  «      «      rdS dS d}t        j                  |«      r%t        |t        t        j                  «      «      sd}t        |t        t        j                  «      «      rd}t        |t        j                  j                  j                  «      rd}|S )zE
        Whether the layer needs to calculate output scales.
        FT)rA   rB   rC   rD   rŠ   rj   r   Úfake_quant_wrap_layersÚis_leaf_layerÚfake_quant_leaf_layersÚquantÚFloatFunctionalLayer)r;   rK   Úflags      r   rŽ   z*ImperativeQuantizeOutputs._is_target_layero  s¼   € ô
 ˜%¤§¡§¡Ô1Øà×Òô ˜e¤U¬5×+GÑ+GÓ%HÔIð ðð ðð ˆÜ×Ñ˜uÔ%¬jØ”5œ×5Ñ5Ó6ô/
ð ˆDä�eœU¤5×#?Ñ#?Ó@ÔAØˆDä�eœVŸY™YŸ_™_×AÑAÔBØˆDàˆr   c                 ó@   ‡‡‡— ˆˆfd„}ˆˆˆfd„} |«         |«        y)z�
        Get all scales from fake ops, save them into the corresponding ops
        and delete all moving_average_abs_max_scale ops.
        c                  óh  •— g } t         j                  dgz   }‰
j                  D ]3  }|j                  D ]"  }|j                  |vsŒ| j                  |«       Œ$ Œ5 | D ]Ô  }t        |«      D ]Ä  }t        j                  |j                  |«      }|€Œ&d|j                  v s|j                  dk(  sŒD|j                  d«      d   }t        j                  ‰|«      }t        j                  |«      }t        ||«      \  }}	|j                  |t        |	«      z   dz   |«       |j                  dd«       ŒÆ ŒÖ y )Nr˜   Úquantize_dequantizeÚOutScaler   Ú
_thresholdÚwith_quant_attrT)r   Ú!fake_quantize_dequantize_op_typesÚblocksÚopsÚtyper   r
   Úfind_previous_opÚblockÚoutputÚload_variable_dataÚfp_numpy_to_naiver	   Ú	_set_attrrl   )Ú
target_opsÚskip_opsrß   ÚopÚin_var_nameÚprevious_opÚ
scale_nameÚin_scaleÚargnameÚindexr›   rÀ   s             €€r   Ú_gather_input_scalezEImperativeQuantizeOutputs._gather_scales.<locals>._gather_input_scale’  s$  ø€ ØˆJÜ×>Ñ>Ø.ðBñ ˆHð !Ÿœ�ØŸ)œ)�BØ—w‘w hÒ.Ø"×)Ñ)¨"Õ-ñ $ð (ó
 !�Ü#:¸2Ö#>�KÜ"'×"8Ñ"8¸¿¹À;Ó"O�Kà"Ñ.Ø-°×1AÑ1AÑAØ&×+Ñ+Ð/MÓMà%0×%7Ñ%7¸
Ó%CÀAÑ%F˜
Ü#(×#;Ñ#;¸EÀ:Ó#N˜Ü#(×#:Ñ#:¸8Ó#D˜Ü)>¸rÀ;Ó)O™˜ ØŸ™Ø#¤c¨%£jÑ0°<Ñ?Àôð Ÿ™Ð%6¸Õ=ñ $?ñ !r   c                  ó`  •— g } ‰j                   D ]4  }|j                  D ]#  }|j                  dk(  sŒ| j                  |«       Œ% Œ6 | D �]b  }|j	                  d«      d   }|j                  d«      d   }|j                  }t        j                  ||«      }t        j                  ||«      }|j                  d«      d   }t        j                  ‰|«      }t        j                  |«      }|j                  dk7  rXt        ||«      }	|	�J|	\  }
}|j                  |
t        |«      z   dz   |«       |j                  d|«       |j                  d	d
«       |D ]T  }|j                  ||«       t!        t#        ‰«      «      D ])  }‰|   j$                  |k(  sŒ|j'                  |«      ‰|<   Œ+ ŒV �Œe y )Nr˜   ÚXr   ÚOutr×   ÚfeedrØ   Úout_thresholdrÙ   T)rÛ   rÜ   rÝ   r   Úinputrà   rß   r   rÞ   Úfind_next_opsrá   râ   r   rã   rl   Ú_rename_inputÚrangeÚlenry   rº   )rä   rß   ræ   rç   Úout_var_namerè   Únext_opsÚout_scale_nameÚ	out_scaleÚresrë   rì   Únext_opÚirÄ   r›   rÀ   s                 €€€r   Ú_gather_output_scalezFImperativeQuantizeOutputs._gather_scales.<locals>._gather_output_scale­  s”  ø€ ØˆJØ Ÿœ�ØŸ)œ)�BØ—w‘wÐ"@Ó@Ø"×)Ñ)¨"Õ-ñ $ð (ô
 !�Ø Ÿh™h s›m¨AÑ.�Ø!Ÿy™y¨Ó/°Ñ2�ØŸ™�Ü#×4Ñ4°U¸KÓH�Ü ×.Ñ.¨u°lÓC�à!#§¡¨:Ó!6°qÑ!9�Ü!×4Ñ4°U¸NÓK�	Ü!×3Ñ3°IÓ>�	à×#Ñ# vÒ-Ü0°¸kÓJ�CØ�Ø),™˜ Ø#×-Ñ-Ø#¤c¨%£jÑ0°<Ñ?Àôð $×-Ñ-¨o¸yÔIØ#×-Ñ-Ð.?ÀÔFã'�GØ×)Ñ)¨,¸ÔDô #¤3 }Ó#5Ö6˜Ø(¨Ñ+×0Ñ0°LÓ@Ø/4¯y©y¸Ó/E˜M¨!Ò,ñ 7ò	  (ñ+ !r   Nr3   )r;   r›   rÀ   rÄ   rí   rÿ   s    ```  r   r¬   z(ImperativeQuantizeOutputs._gather_scalesŒ  s   ú€ õ	>ö6"	FñH 	ÔÙÕr   c                 ó¶   — |j                   D ]J  }|j                  D ]9  }| j                  ||«      sŒ|j                  dd«       |j                  dd«       Œ; ŒL y)z/
        Label the skip quantized ops.
        rs   TrÙ   N)rÛ   rÜ   Ú_is_skip_quant_oprã   )r;   r›   rß   ræ   s       r   r´   z.ImperativeQuantizeOutputs._set_skip_quant_attrÔ  sM   € ð —^”^ˆEØ—i”i�Ø×)Ñ)¨%°Õ4Ø—L‘L ¨tÔ4Ø—L‘LÐ!2°DÕ9ñ  ñ $r   c                 ó°   — g d¢}|j                   |vry|j                  D �cg c]  }t        j                  ||«      ‘Œ }}t	        d„ |D «       «      S c c}w )zÜ
        The input op should be skipped quantization.
        1. the type of input op should be conv2d, depthwise_conv2d or matmul
        2. the previous ops of the input op are not fake_quantize_dequantize ops
        )Úconv2dÚdepthwise_conv2dÚmatmulÚconv2d_transposeFc              3   ó`   K  — | ]&  }|d uxr |j                   t        j                  v–— Œ( y ­wrI   )rÝ   r   rÚ   )rX   ræ   s     r   rY   z>ImperativeQuantizeOutputs._is_skip_quant_op.<locals>.<genexpr>ñ  s<   è ø€ ð 
ñ #�ð �dˆNò GØ—‘œu×FÑFÐFóGá"ùs   ‚,.)rÝ   Úinput_arg_namesr   rÞ   Úany)r;   rß   Úin_opÚtarget_op_typesÚarg_nameÚprevious_opss         r   r  z+ImperativeQuantizeOutputs._is_skip_quant_opÞ  sr   € ò
ˆð �:‰:˜_Ñ,Øð "×1Ò1ó
á1�ô ×"Ñ" 5¨(Õ3Ø1ð 	ð 
ô ñ 
ñ #ó
ó 
ð 	
ùò	
s   ¢A)r'   r&   FrI   )rO   rP   rQ   rR   r5   rE   rJ   rŽ   r¬   r´   r  rS   rT   s   @r   r9   r9   ¾  s0   ø„ ñõ(ò%=óNu$ònò:FòP:ö
r   r9   )r¦   rB   Úpaddle.base.frameworkr   Úpaddle.frameworkr   Úpaddle.nn.quantr   Ú%static.quantization.quantization_passr   r   Ústatic.quantization.utilsr	   r
   r   r   Ú r   r   r©   rª   r   r    r7   r9   r3   r   r   Ú<module>r     sc   ðó 
ã Ý )Ý !Ý (÷÷ó ÷  àÐ Ø"Ð ò3÷w
ñ w
÷tMNñ MN÷`w
ò w
r   