Ë
    �\;j‰Æ  ã                   óÎ   — d dl Z d dlmZ d dlmZ d dlZd dlmZm	Z	 d dl
mZ d dlmZ d dlmZ d dlmZmZ d d	lmZ d
dlmZ  G d„ de«      Zd„ Z G d„ d«      Z G d„ de«      Zy)é    N)Údefaultdict)ÚEnum)Ú_C_opsÚ_legacy_C_ops)Úcore)Ú
check_type)Úto_variable)Ú_dygraph_tracerÚdygraph_only)Úin_dynamic_modeé   )Úamp_global_statec                   ó   — e Zd ZdZdZdZy)ÚOptimizerStater   r   é   N)Ú__name__Ú
__module__Ú__qualname__ÚINITÚUNSCALEDÚSTEPPED© ó    ú_G:\00. PROJECTS\API\Inventory\templateJSON\kerjaOCR\Lib\site-packages\paddle/amp/grad_scaler.pyr   r      s   „ Ø€DØ€HØ�Gr   r   c                  ó&   — dt         j                  iS )NÚstate)r   r   r   r   r   Ú_refresh_optimizer_stater   %   s   € Ø”^×(Ñ(Ð)Ð)r   c                   óœ   — e Zd ZdZe	 	 	 	 	 	 	 dd„«       Zd„ Zd„ Zd„ Zd„ Z	d„ Z
d„ Zd	„ Zd
„ Zd„ Zd„ Zd„ Zd„ Zd„ Zd„ Zd„ Zd„ Zd„ Zd„ Zy)Ú	AmpScalera¾	  
    AmpScaler is used for Auto-Mixed-Precision training/inferring in imperative
    mode. It controls the scaling of loss, helps avoiding numerical overflow.
    The object of this class has seventeen methods `scale()`, `unscale_()`, `minimize()` and `get`/`set` api of parameters.

    `scale()` is used to multiply the loss by a scale ratio.
    `unscale_()` is used to unscale the gradients of parameters, multiplies the gradients of parameters by 1/(scale ratio)
    `minimize()` is similar as `optimizer.minimize()`, performs parameters updating, and it will update the loss_scaling.

    Commonly, it is used together with `amp_guard` to achieve Auto-Mixed-Precision in
    imperative mode.

    Args:
        enable(bool, optional): Enable loss scaling or not. Default is True.
        init_loss_scaling (float, optional): The initial loss scaling factor. Default is 2**15.
        incr_ratio(float, optional): The multiplier to use when increasing the loss
                        scaling. Default is 2.0.
        decr_ratio(float, optional): The less-than-one-multiplier to use when decreasing
                        the loss scaling. Default is 0.5.
        incr_every_n_steps(int, optional): Increases loss scaling every n consecutive
                                steps with finite gradients. Default is 1000.
        decr_every_n_nan_or_inf(int, optional): Decreases loss scaling every n
                                    accumulated steps with nan or inf gradients. Default is 2.
        use_dynamic_loss_scaling(bool, optional): Whether to use dynamic loss scaling. If False, fixed loss_scaling is used. If True, the loss scaling is updated dynamicly. Default is True.
    Returns:
        An AmpScaler object.

    Examples:

        .. code-block:: python

            >>> import numpy as np
            >>> import paddle

            >>> data = np.random.uniform(-1, 1, [10, 3, 32, 32]).astype('float32')
            >>> model = paddle.nn.Conv2D(3, 2, 3)
            >>> optimizer = paddle.optimizer.SGD(
            ...         learning_rate=0.01, parameters=model.parameters())
            >>> scaler = paddle.amp.AmpScaler(init_loss_scaling=1024)
            >>> data = paddle.to_tensor(data)
            >>> with paddle.amp.amp_guard():
            ...     conv = model(data)
            ...     loss = paddle.mean(conv)
            ...     scaled = scaler.scale(loss)
            ...     scaled.backward()
            ...     scaler.minimize(optimizer, scaled)
    c                 ó:  — t        «       }|st        d«      ‚|rr|j                  j                  «       sX|j                  j	                  «       s>|j                  j                  «       s$t        j                  d|j                  z  «       d}|| _        | j                  �rü|dkD  sJ d«       ‚|dk  sJ d«       ‚|| _	        || _
        || _        || _        || _        d| _        d| _        || _        t#        t%        j&                  dg«      j)                  t$        j*                  «      «      | _        t#        t%        j&                  dg«      j)                  t$        j*                  «      «      | _        t#        t%        j&                  dg«      j)                  t$        j*                  «      «      | _        t#        t%        j&                  dg«      j)                  t$        j*                  «      «      | _        t#        t%        j&                  dg«      j)                  t$        j*                  «      «      | _        t#        t%        j&                  | j                  g«      j)                  t$        j6                  «      «      | _        d | _        t=        t>        «      | _         y y )Nz;current_tracer is None, maybe it is not in imperative mode.zqAmpScaler can only be enabled on CUDAPlace, XPUPlace and CustomPlace, current place is %s, so it makes no effect.Fç      ð?zThe incr_ratio must be > 1.0.zThe decr_ratio must be < 1.0.r   )!r
   Ú
ValueErrorÚ_expected_placeÚis_gpu_placeÚis_xpu_placeÚis_custom_placeÚwarningsÚwarnÚ_enableÚ_init_loss_scalingÚ_incr_ratioÚ_decr_ratioÚ_incr_every_n_stepsÚ_decr_every_n_nan_or_infÚ_incr_countÚ_decr_countÚ_use_dynamic_loss_scalingr	   ÚnpÚarrayÚastypeÚbool_Ú
_found_infÚ_temp_found_inf_value_falseÚ_temp_found_inf_fp16Ú_temp_found_inf_bf16Ú_temp_found_inf_fp32Úfloat32Ú_scaleÚ_cache_founf_infr   r   Ú_optimizer_states)	ÚselfÚenableÚinit_loss_scalingÚ
incr_ratioÚ
decr_ratioÚincr_every_n_stepsÚdecr_every_n_nan_or_infÚuse_dynamic_loss_scalingÚtracers	            r   Ú__init__zAmpScaler.__init__Z   s  € ô !Ó"ˆÙÜØMóð ñ Ø×"Ñ"×/Ñ/Ô1Ø×%Ñ%×2Ñ2Ô4Ø×%Ñ%×5Ñ5Ô7ä�M‰Mð DØ×(Ñ(ñ)ôð ˆFàˆŒà�<‹<Ø Ò#ÐDÐ%DÓDÐ#Ø Ò#ÐDÐ%DÓDÐ#à&7ˆDÔ#Ø)ˆDÔØ)ˆDÔØ'9ˆDÔ$Ø,CˆDÔ)Ø ˆDÔØ ˆDÔØ-EˆDÔ*ä)¬"¯(©(°A°3«-×*>Ñ*>¼r¿x¹xÓ*HÓIˆDŒOÜ/:Ü—‘˜!˜“×$Ñ$¤R§X¡XÓ.ó0ˆDÔ,ô )4Ü—‘˜!˜“×$Ñ$¤R§X¡XÓ.ó)ˆDÔ%ô )4Ü—‘˜!˜“×$Ñ$¤R§X¡XÓ.ó)ˆDÔ%ô )4Ü—‘˜!˜“×$Ñ$¤R§X¡XÓ.ó)ˆDÔ%ô &Ü—‘˜$×1Ñ1Ð2Ó3×:Ñ:¼2¿:¹:ÓFóˆDŒKð %)ˆDÔ!Ü%0Ô1IÓ%JˆDÕ"ð= r   c                 óV  — t        |dt        j                  j                  d«       | j                  r[t        «       j                  dk7  rD| j                  r8d| _        d| _        t        j                  dt        «       j                  z  «       | j                  s|S || j                  z  S )al  
        Multiplies a Tensor by the scale factor and returns scaled outputs.
        If this instance of :class:`AmpScaler` is not enabled, output are returned unmodified.

        Args:
            var (Tensor):  The Tensor to scale.
        Returns:
            The scaled Tensor or original Tensor.

        Examples:

            .. code-block:: python

                >>> import numpy as np
                >>> import paddle

                >>> data = np.random.uniform(-1, 1, [10, 3, 32, 32]).astype('float32')
                >>> model = paddle.nn.Conv2D(3, 2, 3)
                >>> optimizer = paddle.optimizer.SGD(
                ...         learning_rate=0.01, parameters=model.parameters())
                >>> scaler = paddle.amp.AmpScaler(init_loss_scaling=1024)
                >>> data = paddle.to_tensor(data)
                >>> with paddle.amp.amp_guard():
                ...     conv = model(data)
                ...     loss = paddle.mean(conv)
                ...     scaled = scaler.scale(loss)
                ...     scaled.backward()
                ...     scaler.minimize(optimizer, scaled)
        ÚvarzAmpScaler.scale()Úfloat16Fz^It is not recommended to use dynamic loss scaling for %s, so GradScaler is disable by default.)r   r   ÚeagerÚTensorr)   r   Ú	amp_dtyper1   r'   r(   r<   )r?   rJ   s     r   ÚscalezAmpScaler.scale˜   sŒ   € ô< 	�3˜œtŸz™z×0Ñ0Ð2EÔFð �LŠLÜ Ó"×,Ñ,°	Ò9Ø×.Ò.à ˆDŒLØ-2ˆDÔ*Ü�M‰MØpÜ#Ó%×/Ñ/ñ1ôð
 �|Š|ØˆJà�T—[‘[Ñ Ð r   c                 ó4  — | j                   s |j                  |i |¤ŽS | j                  t        |«         }|d   t        j
                  u r| j                  |«       d\  }}t        |d«      rH|j                  d| j                  «        |j                  |i |¤Ž\  }}|j                  d«      | _        n0| j                  rd| _        n |j                  |i |¤Ž\  }}d| _        | j                  r| j                  «        t        t        «      | _        ||fS )aò  
        This function is similar as `Optimizer.minimize()`, which performs parameters updating.

        If the scaled gradients of parameters contains NAN or INF, the parameters updating is skipped.
        Otherwise, if `unscale_()` has not been called, it first unscales the scaled gradients of parameters, then updates the parameters.

        Finally, the loss scaling ratio is updated.

        Args:
            optimizer(Optimizer):  The optimizer used to update parameters.
            args:  Arguments, which will be forward to `optimizer.minimize()`.
            kwargs: Keyword arguments, which will be forward to `Optimizer.minimize()`.

        Examples:

            .. code-block:: python

                >>> import numpy as np
                >>> import paddle

                >>> data = np.random.uniform(-1, 1, [10, 3, 32, 32]).astype('float32')
                >>> model = paddle.nn.Conv2D(3, 2, 3)
                >>> optimizer = paddle.optimizer.SGD(
                ...     learning_rate=0.01,
                ...     parameters=model.parameters()
                ... )
                >>> scaler = paddle.amp.AmpScaler(init_loss_scaling=1024)
                >>> data = paddle.to_tensor(data)
                >>> with paddle.amp.amp_guard():
                ...     conv = model(data)
                ...     loss = paddle.mean(conv)
                ...     scaled = scaler.scale(loss)
                ...     scaled.backward()
                ...     scaler.minimize(optimizer, scaled)
        r   )NNÚ_set_auxiliary_varÚ	found_infTF)r)   Úminimizer>   Úidr   r   Ú_unscaleÚhasattrrQ   r6   Ú_get_auxiliary_varr=   r1   Ú_updater   r   )r?   Ú	optimizerÚargsÚkwargsÚoptimizer_stateÚoptimize_opsÚparams_gradss          r   rS   zAmpScaler.minimizeÉ   s  € ðH �|Š|Ø%�9×%Ñ% tÐ6¨vÑ6Ð6à×0Ñ0´°I³Ñ?ˆð ˜7Ñ#¤~×':Ñ':Ñ:Ø�M‰M˜)Ô$à%1Ñ"ˆ�lä�9Ð2Ô3Ø×(Ñ(¨°d·o±oÔFØ);¨×);Ñ);¸TÐ)LÀVÑ)LÑ&ˆL˜,Ø$-×$@Ñ$@ÀÓ$MˆDÕ!à�ŠØ(,�Õ%à-?¨Y×-?Ñ-?ÀÐ-PÈÑ-PÑ*�˜lØ(-�Ô%à×)Ò)à�L‰LŒNä!,Ô-EÓ!FˆÔà˜\Ð)Ð)r   c                 óÄ  — | j                   sy| j                  t        |«         }|d   t        j                  u rt        d«      ‚|d   t        j                  u rt        d«      ‚t        |dd«      �rTt        |j                  d   t        «      �r6g }g }g }g }|j                  D �]  }|d   D �]  }|j                  «       €Œ|j                  |j                  «       «       |j                  «       j                  t        j                  j                   j"                  k(  r |j                  |j                  «       «       Œ“|j                  «       j                  t        j                  j                   j$                  k(  r |j                  |j                  «       «       Œò|j                  |j                  «       «       �Œ �Œ �n/t'        «       r.t        j(                  j+                  |j,                  «      \  }}}n÷|j,                  D �cg c]"  }|j                  «       �|j                  «       ‘Œ$ }}|D �cg c]5  }|j                  t        j                  j                   j"                  k(  r|‘Œ7 }}|D �cg c]5  }|j                  t        j                  j                   j$                  k(  r|‘Œ7 }}|D �cg c]5  }|j                  t        j                  j                   j.                  k(  r|‘Œ7 }}| j0                  | _        t5        |«      r[t7        j8                  || j:                  || j<                  «       t?        j@                  | j2                  | j<                  «      | _        t5        |«      r[t7        j8                  || j:                  || jB                  «       t?        j@                  | j2                  | jB                  «      | _        t5        |«      r[t7        j8                  || j:                  || jD                  «       t?        j@                  | j2                  | jD                  «      | _        t        j                  |d<   yc c}w c c}w c c}w c c}w )a  
        Unscale the gradients of parameters, multiplies the gradients of parameters by 1/(loss scaling ratio).
        If this instance of :class:`GradScaler` is not enabled, output are returned unmodified.
        Args:
            optimizer(Optimizer):  The optimizer used to update parameters.
        Returns:
            The unscaled parameters or original parameters.
        Nr   zMunscale_() has already been called on this optimizer since the last update().z(unscale_() is being called after step().Ú_param_groupsr   Úparams)#r)   r>   rT   r   r   ÚRuntimeErrorr   ÚgetattrÚ
isinstancer`   ÚdictÚ
_grad_ivarÚappendÚdtyper   ÚVarDescÚVarTypeÚFP16ÚBF16r   rL   Úget_grads_listsÚ_parameter_listÚFP32r7   r6   Úlenr   Úcheck_finite_and_unscaler<   r8   r   Ú
bitwise_orr9   r:   )	r?   rY   r\   Úparam_gradsÚparam_grads_fp16Úparam_grads_bf16Úparam_grads_fp32ÚgroupÚparams	            r   rU   zAmpScaler._unscale  s×  € ð �|Š|Øà×0Ñ0´°I³Ñ?ˆà˜7Ñ#¤~×'>Ñ'>Ñ>ÜØ_óð ð ˜WÑ%¬×)?Ñ)?Ñ?ÜÐIÓJÐJä�9˜o¨tÕ4¼Ø×#Ñ# AÑ&¬õ:
ð ˆKØ!ÐØ!ÐØ!ÐØ"×0Õ0�Ø" 8�_�EØ×'Ñ'Ó)Ñ5Ø#×*Ñ*¨5×+;Ñ+;Ó+=Ô>à!×,Ñ,Ó.×4Ñ4Ü#Ÿ|™|×3Ñ3×8Ñ8ò9ð -×3Ñ3°E×4DÑ4DÓ4FÕGà!×,Ñ,Ó.×4Ñ4Ü#Ÿ|™|×3Ñ3×8Ñ8ò9ð -×3Ñ3°E×4DÑ4DÓ4FÕGà,×3Ñ3°E×4DÑ4DÓ4FÖGò -ò 1ô" Ô ô —J‘J×.Ñ.¨y×/HÑ/HÓIñ	Ø$Ø$Ù$ð "+×!:Ò!:óá!:˜Ø×'Ñ'Ó)Ð5ð ×$Ñ$Õ&Ø!:ð ð ñ "-ó$á!,˜Ø—{‘{¤d§l¡l×&:Ñ&:×&?Ñ&?Ò?ò Ø!,ð !ð $ñ "-ó$á!,˜Ø—{‘{¤d§l¡l×&:Ñ&:×&?Ñ&?Ò?ò Ø!,ð !ð $ñ "-ó$á!,˜Ø—{‘{¤d§l¡l×&:Ñ&:×&?Ñ&?Ò?ò Ø!,ð !ð $ð
 ×:Ñ:ˆŒÜÐÔ Ü×2Ñ2Ø Ø—‘Ø Ø×)Ñ)ô	ô %×/Ñ/Ø—‘ ×!:Ñ!:óˆDŒOô ÐÔ Ü×2Ñ2Ø Ø—‘Ø Ø×)Ñ)ô	ô %×/Ñ/Ø—‘ ×!:Ñ!:óˆDŒOô ÐÔ Ü×2Ñ2Ø Ø—‘Ø Ø×)Ñ)ô	ô %×/Ñ/Ø—‘ ×!:Ñ!:óˆDŒOô $2×#:Ñ#:ˆ˜Ò ùòiùò
$ùò
$ùò
$s   È'QÈ;:QÉ;:QÊ;:Qc           	      óF  — | j                   sy| j                  r¯d| _        | j                  dz   | _        | j                  | j                  k(  rzt        dj                  t        | j                  «      t        | j                  «      t        | j                  «      «      «       | j                  | j                  z  | _        d| _        yd| _        | j                  dz   | _        | j                  | j                  k(  r%| j                  | j                  z  | _        d| _        y)z+
        Updates the loss_scaling.
        Nr   r   z:Found inf or nan, current scale is: {}, decrease to: {}*{})r)   r=   r/   r0   r.   ÚprintÚformatÚfloatr<   r,   r-   r+   ©r?   s    r   rX   zAmpScaler._updatey  sù   € ð �|Š|Øà× Ò Ø ˆDÔØ#×/Ñ/°!Ñ3ˆDÔØ×Ñ 4×#@Ñ#@Ò@ÜØP×WÑWÜ˜dŸk™kÓ*Ü˜dŸk™kÓ*Ü˜d×.Ñ.Ó/óôð #Ÿk™k¨D×,<Ñ,<Ñ<�”Ø#$�Ô ð 	ð  !ˆDÔØ#×/Ñ/°!Ñ3ˆDÔØ×Ñ 4×#;Ñ#;Ò;Ø"Ÿk™k¨D×,<Ñ,<Ñ<�”Ø#$�Ô àr   c                 ó   — | j                   S )z„
        Enable loss scaling or not.

        Returns:
            bool: enable loss scaling return True else return False.
        )r)   r}   s    r   Ú	is_enablezAmpScaler.is_enable–  s   € ð �|‰|Ðr   c                 ó   — | j                   S )z¼
        Whether to use dynamic loss scaling.

        Returns:
            bool: if fixed loss_scaling is used return False, if the loss scaling is updated dynamicly return true.
        )r1   r}   s    r   Úis_use_dynamic_loss_scalingz%AmpScaler.is_use_dynamic_loss_scalingŸ  s   € ð ×-Ñ-Ð-r   c                 ó   — | j                   S )z
        Return the initial loss scaling factor.

        Reurns:
            float:  the initial loss scaling factor.
        )r*   r}   s    r   Úget_init_loss_scalingzAmpScaler.get_init_loss_scaling¨  s   € ð ×&Ñ&Ð&r   c                 ó¨   — || _         t        t        j                  | j                   g«      j	                  t        j
                  «      «      | _        y)zÐ
        Set the initial loss scaling factor by `new_init_loss_scaling`.

        Args:
            new_init_loss_scaling(int):  The new_init_loss_scaling used to update initial loss scaling factor.s
        N)r*   r	   r2   r3   r4   r;   r<   )r?   Únew_init_loss_scalings     r   Úset_init_loss_scalingzAmpScaler.set_init_loss_scaling±  s<   € ð #8ˆÔÜ!Ü�H‰H�d×-Ñ-Ð.Ó/×6Ñ6´r·z±zÓBó
ˆ�r   c                 ó   — | j                   S )z­
        Return the multiplier to use when increasing the loss scaling.

        Reurns:
            float:  the multiplier to use when increasing the loss scaling.
        ©r+   r}   s    r   Úget_incr_ratiozAmpScaler.get_incr_ratio½  ó   € ð ×ÑÐr   c                 ó*   — |dkD  sJ d«       ‚|| _         y)a  
        Set the multiplier to use when increasing the loss scaling by `new_incr_ratio`, `new_incr_ratio` should > 1.0.

        Args:
            new_incr_ratio(float):  The new_incr_ratio used to update the multiplier to use when increasing the loss scaling.
        r!   z!The new_incr_ratio must be > 1.0.Nrˆ   )r?   Únew_incr_ratios     r   Úset_incr_ratiozAmpScaler.set_incr_ratioÆ  ó    € ð  Ò#ÐHÐ%HÓHÐ#Ø)ˆÕr   c                 ó   — | j                   S )zÆ
        Get the less-than-one-multiplier to use when decreasing the loss scaling.

        Reurns:
            float:  the less-than-one-multiplier to use when decreasing the loss scaling.
        ©r,   r}   s    r   Úget_decr_ratiozAmpScaler.get_decr_ratioÐ  rŠ   r   c                 ó*   — |dk  sJ d«       ‚|| _         y)a)  
        Set the less-than-one-multiplier to use when decreasing the loss scaling by `new_incr_ratio`, `new_decr_ratio` should < 1.0.

        Args:
            new_decr_ratio(float):  The new_decr_ratio used to update the less-than-one-multiplier to use when decreasing the loss scaling.
        r!   z!The new_decr_ratio must be < 1.0.Nr�   )r?   Únew_decr_ratios     r   Úset_decr_ratiozAmpScaler.set_decr_ratioÙ  rŽ   r   c                 ó   — | j                   S )a  
        Return the num `n`, `n` represent increases loss scaling every `n` consecutive steps with finite gradients.

        Reurns:
            int:  the num `n`, `n` represent increases loss scaling every `n` consecutive steps with finite gradients.
        ©r-   r}   s    r   Úget_incr_every_n_stepsz AmpScaler.get_incr_every_n_stepsã  s   € ð ×'Ñ'Ð'r   c                 ó   — || _         y)a^  
        Set the num `n` by `new_incr_every_n_steps`, `n` represent increases loss scaling every `n` consecutive steps with finite gradients.

        Args:
            new_incr_every_n_steps(int):  The new_incr_every_n_steps used to update the num `n`, `n` represent increases loss scaling every `n` consecutive steps with finite gradients.
        Nr–   )r?   Únew_incr_every_n_stepss     r   Úset_incr_every_n_stepsz AmpScaler.set_incr_every_n_stepsì  s   € ð $:ˆÕ r   c                 ó   — | j                   S )a  
        Return the num `n`, `n` represent decreases loss scaling every `n` accumulated steps with nan or inf gradients.

        Reurns:
            int:  the num `n`, `n` represent decreases loss scaling every `n` accumulated steps with nan or inf gradients.
        ©r.   r}   s    r   Úget_decr_every_n_nan_or_infz%AmpScaler.get_decr_every_n_nan_or_infõ  s   € ð ×,Ñ,Ð,r   c                 ó   — || _         y)au  
        Set the num `n` by `new_decr_every_n_nan_or_inf`, `n` represent decreases loss scaling every `n` accumulated steps with nan or inf gradients.

        Args:
            new_decr_every_n_nan_or_inf(int):  The new_decr_every_n_nan_or_inf used to update the num `n`, `n` represent decreases loss scaling every `n` accumulated steps with nan or inf gradients.
        Nrœ   )r?   Únew_decr_every_n_nan_or_infs     r   Úset_decr_every_n_nan_or_infz%AmpScaler.set_decr_every_n_nan_or_infþ  s   € ð )DˆÕ%r   c           	      óð   — | j                   ri| j                  j                  «       | j                  | j                  | j
                  | j                  | j                  | j                  | j                  dœS i S )aÖ  
        Returns the state of the scaler as a `dict`, If this instance is not enabled, returns an empty dict.

        Reurns:
            A dict of scaler includes:
            scale (tensor): The loss scaling factor.
            incr_ratio(float): The multiplier to use when increasing the loss scaling.
            decr_ratio(float): The less-than-one-multiplier to use when decreasing the loss scaling.
            incr_every_n_steps(int): Increases loss scaling every n consecutive steps with finite gradients.
            decr_every_n_nan_or_inf(int): Decreases loss scaling every n accumulated steps with nan or inf gradients.
            incr_count(int): The number of recent consecutive unskipped steps.
            decr_count(int): The number of recent consecutive skipped steps.
            use_dynamic_loss_scaling(bool): Whether to use dynamic loss scaling. If False, fixed loss_scaling is used. If True, the loss scaling is updated dynamicly. Default is True.
        )rO   rB   rC   rD   rE   Ú
incr_countÚ
decr_countrF   )
r)   r<   Únumpyr+   r,   r-   r.   r/   r0   r1   r}   s    r   Ú
state_dictzAmpScaler.state_dict  sp   € ð4 �|Š|ð Ÿ™×*Ñ*Ó,Ø"×.Ñ.Ø"×.Ñ.Ø&*×&>Ñ&>Ø+/×+HÑ+HØ"×.Ñ.Ø"×.Ñ.Ø,0×,JÑ,Jñ	ð	
ð ð	
r   c                 óŒ  — | j                   syt        |«      dk(  rt        d«      ‚|d   d   | _        t	        t        j                  | j                  g«      j                  t
        j                  «      «      | _	        |d   | _
        |d   | _        |d   | _        |d   | _        |d	   | _        |d
   | _        |d   | _        y)zª
        Loads the scaler state.

        Args:
           state_dict(dict): scaler state.  Should be an object returned from a call to `AmpScaler.state_dict()`.
        Nr   zdThe input state dict is empty, possibly because it was saved from a disabled instance of GradScaler.rO   rB   rC   rD   rE   r¢   r£   rF   )r)   rp   rb   r*   r	   r2   r3   r4   r;   r<   r+   r,   r-   r.   r/   r0   r1   )r?   r¥   s     r   Úload_state_dictzAmpScaler.load_state_dict%  sÐ   € ð �|Š|Øäˆz‹?˜aÒÜð:óð ð
 #-¨WÑ"5°aÑ"8ˆÔÜ!Ü�H‰H�d×-Ñ-Ð.Ó/×6Ñ6´r·z±zÓBó
ˆŒð & lÑ3ˆÔØ% lÑ3ˆÔØ#-Ð.BÑ#CˆÔ Ø(2Ð3LÑ(MˆÔ%Ø% lÑ3ˆÔØ% lÑ3ˆÔØ)3Ð4NÑ)OˆÕ&r   N)Tg      à@ç       @ç      à?iè  r   T)r   r   r   Ú__doc__r   rH   rO   rS   rU   rX   r   r�   rƒ   r†   r‰   r�   r‘   r”   r—   rš   r�   r    r¥   r§   r   r   r   r   r   )   s—   „ ñ.ð` ð Ø!ØØØØ !Ø!%ò;Kó ð;Kòz/!òb@*òDl;ò\ò:ò.ò'ò

ò ò*ò ò*ò(ò:ò-òDò
ó<Pr   r   c                   óè   ‡ — e Zd ZdZ	 	 	 	 	 	 	 dˆ fd„	Zˆ fd„Zˆ fd„Zd„ Zd„ Zˆ fd„Z	ˆ fd„Z
ˆ fd	„Zˆ fd
„Zˆ fd„Zˆ fd„Zˆ fd„Zˆ fd„Zˆ fd„Zˆ fd„Zˆ fd„Zˆ fd„Zˆ fd„Zˆ fd„Zˆ fd„Zˆ xZS )Ú
GradScalerak
  
    GradScaler is used for Auto-Mixed-Precision training in dynamic graph mode.
    It controls the scaling of loss, helps avoiding numerical overflow.
    The object of this class has nineteen methods `scale()`, `unscale_()`, `minimize()`, `step()`, `update()` and `get`/`set` api of parameters.

    `scale()` is used to multiply the loss by a scale ratio.
    `unscale_()` is used to unscale the gradients of parameters, multiplies the gradients of parameters by 1/(scale ratio)
    `minimize()` is similar as `optimizer.minimize()`, performs parameters updating, and it will update the loss_scaling, it equal to `step()` + `update()`.
    `step()` is similar as `optimizer.step()`, which performs parameters updating.
    `update` is used to update the loss_scaling.


    Commonly, it is used together with `paddle.amp.auto_cast` to achieve Auto-Mixed-Precision in
    dynamic graph mode.

    Args:
        enable(bool, optional): Enable loss scaling or not. Default is True.
        init_loss_scaling (float, optional): The initial loss scaling factor. Default is 65536.0.
        incr_ratio(float, optional): The multiplier to use when increasing the loss
                        scaling. Default is 2.0.
        decr_ratio(float, optional): The less-than-one-multiplier to use when decreasing
                        the loss scaling. Default is 0.5.
        incr_every_n_steps(int, optional): Increases loss scaling every n consecutive
                                steps with finite gradients. Default is 2000.
        decr_every_n_nan_or_inf(int, optional): Decreases loss scaling every n
                                    accumulated steps with nan or inf gradients. Default is 1.
        use_dynamic_loss_scaling(bool, optional): Whether to use dynamic loss scaling. If False, fixed loss_scaling is used. If True, the loss scaling is updated dynamicly. Default is True.
    Returns:
        An GradScaler object.

    Examples:

        .. code-block:: python

            >>> import paddle

            >>> model = paddle.nn.Conv2D(3, 2, 3, bias_attr=True)
            >>> optimizer = paddle.optimizer.SGD(learning_rate=0.01, parameters=model.parameters())
            >>> scaler = paddle.amp.GradScaler(init_loss_scaling=1024)
            >>> data = paddle.rand([10, 3, 32, 32])

            >>> with paddle.amp.auto_cast():
            ...     conv = model(data)
            ...     loss = paddle.mean(conv)

            >>> scaled = scaler.scale(loss)  # scale the loss
            >>> scaled.backward()            # do backward
            >>> scaler.minimize(optimizer, scaled)  # update parameters
            >>> optimizer.clear_grad()
    c           	      ó0   •— t         ‰| �  |||||||«       y )N)ÚsuperrH   )	r?   r@   rA   rB   rC   rD   rE   rF   Ú	__class__s	           €r   rH   zGradScaler.__init__v  s'   ø€ ô 	‰ÑØØØØØØ#Ø$õ	
r   c                 ó"   •— t         ‰| �  |«      S )aJ  
        Multiplies a Tensor by the scale factor and returns scaled outputs.
        If this instance of :class:`GradScaler` is not enabled, output are returned unmodified.

        Args:
            var (Tensor):  The tensor to scale.
        Returns:
            The scaled tensor or original tensor.

        Examples:

            .. code-block:: python

                >>> import paddle

                >>> model = paddle.nn.Conv2D(3, 2, 3, bias_attr=True)
                >>> optimizer = paddle.optimizer.SGD(learning_rate=0.01, parameters=model.parameters())
                >>> scaler = paddle.amp.GradScaler(init_loss_scaling=1024)
                >>> data = paddle.rand([10, 3, 32, 32])

                >>> with paddle.amp.auto_cast():
                ...     conv = model(data)
                ...     loss = paddle.mean(conv)

                >>> scaled = scaler.scale(loss)  # scale the loss
                >>> scaled.backward()            # do backward
                >>> scaler.minimize(optimizer, scaled)  # update parameters
                >>> optimizer.clear_grad()
        )r®   rO   )r?   rJ   r¯   s     €r   rO   zGradScaler.scaleŠ  s   ø€ ô< ‰w‰}˜SÓ!Ð!r   c                 ó*   •— t        ‰| �  |g|¢­i |¤ŽS )a¦  
        This function is similar as `optimizer.minimize()`, which performs parameters updating.

        If the scaled gradients of parameters contains NAN or INF, the parameters updating is skipped.
        Otherwise, if `unscale_()` has not been called, it first unscales the scaled gradients of parameters, then updates the parameters.

        Finally, the loss scaling ratio is updated.

        Args:
            optimizer(Optimizer):  The optimizer used to update parameters.
            args:  Arguments, which will be forward to `optimizer.minimize()`.
            kwargs: Keyword arguments, which will be forward to `optimizer.minimize()`.

        Examples:

            .. code-block:: python

                >>> import paddle

                >>> model = paddle.nn.Conv2D(3, 2, 3, bias_attr=True)
                >>> optimizer = paddle.optimizer.SGD(learning_rate=0.01, parameters=model.parameters())
                >>> scaler = paddle.amp.GradScaler(init_loss_scaling=1024)
                >>> data = paddle.rand([10, 3, 32, 32])

                >>> with paddle.amp.auto_cast():
                ...     conv = model(data)
                ...     loss = paddle.mean(conv)

                >>> scaled = scaler.scale(loss)  # scale the loss
                >>> scaled.backward()            # do backward
                >>> scaler.minimize(optimizer, scaled)  # update parameters
                >>> optimizer.clear_grad()
        )r®   rS   )r?   rY   rZ   r[   r¯   s       €r   rS   zGradScaler.minimizeª  s    ø€ ôD ‰wÑ 	Ð;¨DÒ;°FÑ;Ð;r   c                 óT  — | j                   s|j                  «       S | j                  t        |«         }|d   t        j
                  u rt        d«      ‚|d   t        j                  u r| j                  |«       t        |d«      rC|j                  d| j                  «       |j                  «        |j                  d«      | _        n+| j                  rd| _        n|j                  «        d| _        t        j
                  |d<   | j                  st        t         «      | _        yy)at  
        This function is similar as `optimizer.step()`, which performs parameters updating.

        If the scaled gradients of parameters contains NAN or INF, the parameters updating is skipped.
        Otherwise, if `unscale_()` has not been called, it first unscales the scaled gradients of parameters, then updates the parameters.

        Args:
            optimizer(Optimizer):  The optimizer used to update parameters.

        Examples:

            .. code-block:: python

                >>> # doctest: +REQUIRES(env:GPU)
                >>> import paddle
                >>> paddle.device.set_device('gpu')

                >>> model = paddle.nn.Conv2D(3, 2, 3, bias_attr=True)
                >>> optimizer = paddle.optimizer.SGD(learning_rate=0.01, parameters=model.parameters())
                >>> scaler = paddle.amp.GradScaler(init_loss_scaling=1024)
                >>> data = paddle.rand([10, 3, 32, 32])
                >>> with paddle.amp.auto_cast():
                ...     conv = model(data)
                ...     loss = paddle.mean(conv)
                >>> scaled = scaler.scale(loss)  # scale the loss
                >>> scaled.backward()            # do backward
                >>> scaler.step(optimizer)       # update parameters
                >>> scaler.update()              # update the loss scaling ratio
                >>> optimizer.clear_grad()
        r   z7step() has already been called since the last update().rQ   rR   TFN)r)   Ústepr>   rT   r   r   rb   r   rU   rV   rQ   r6   rW   r=   r1   r   r   )r?   rY   r\   s      r   r³   zGradScaler.stepÎ  sù   € ð> �|Š|Ø—>‘>Ó#Ð#à×0Ñ0´°I³Ñ?ˆØ˜7Ñ#¤~×'=Ñ'=Ñ=ÜØIóð ð
 ˜7Ñ#¤~×':Ñ':Ñ:Ø�M‰M˜)Ô$ä�9Ð2Ô3Ø×(Ñ(¨°d·o±oÔFØ�N‰NÔØ$-×$@Ñ$@ÀÓ$MˆDÕ!à�ŠØ(,�Õ%à—‘Ô Ø(-�Ô%ä#1×#9Ñ#9ˆ˜Ñ à×-Ò-Ü%0Ô1IÓ%JˆDÕ"ð .r   c                 ó~   — | j                   sy| j                  r$| j                  «        t        t        «      | _        y)aø  
        Updates the loss_scaling.

        Examples:

            .. code-block:: python

                >>> # doctest: +REQUIRES(env:GPU)
                >>> import paddle

                >>> paddle.device.set_device('gpu')
                >>> model = paddle.nn.Conv2D(3, 2, 3, bias_attr=True)
                >>> optimizer = paddle.optimizer.SGD(learning_rate=0.01, parameters=model.parameters())
                >>> scaler = paddle.amp.GradScaler(init_loss_scaling=1024)
                >>> data = paddle.rand([10, 3, 32, 32])
                >>> with paddle.amp.auto_cast():
                ...     conv = model(data)
                ...     loss = paddle.mean(conv)
                >>> scaled = scaler.scale(loss)     # scale the loss
                >>> scaled.backward()               # do backward
                >>> scaler.step(optimizer)          # update parameters
                >>> scaler.update()                 # update the loss scaling ratio
                >>> optimizer.clear_grad()
        N)r)   r1   rX   r   r   r>   r}   s    r   ÚupdatezGradScaler.update
  s1   € ð2 �|Š|ØØ×)Ò)Ø�L‰LŒNÜ%0Ô1IÓ%JˆDÔ"Ør   c                 ó"   •— t         ‰| �  |«      S )aE  
        Unscale the gradients of parameters, multiplies the gradients of parameters by 1/(loss scaling ratio).
        If this instance of :class:`GradScaler` is not enabled, output are returned unmodified.

        Args:
            optimizer(Optimizer):  The optimizer used to update parameters.

        Returns:
            The unscaled parameters or original parameters.

        Examples:

            .. code-block:: python

                >>> # doctest: +REQUIRES(env:GPU)
                >>> import paddle

                >>> paddle.device.set_device('gpu')
                >>> model = paddle.nn.Conv2D(3, 2, 3, bias_attr=True)
                >>> optimizer = paddle.optimizer.SGD(learning_rate=0.01, parameters=model.parameters())
                >>> scaler = paddle.amp.GradScaler(init_loss_scaling=1024)
                >>> data = paddle.rand([10, 3, 32, 32])
                >>> with paddle.amp.auto_cast():
                ...     conv = model(data)
                ...     loss = paddle.mean(conv)
                >>> scaled = scaler.scale(loss)  # scale the loss
                >>> scaled.backward()            # do backward
                >>> scaler.unscale_(optimizer)    # unscale the parameter
                >>> scaler.step(optimizer)
                >>> scaler.update()
                >>> optimizer.clear_grad()
        )r®   rU   )r?   rY   r¯   s     €r   Úunscale_zGradScaler.unscale_*  s   ø€ ôB ‰wÑ 	Ó*Ð*r   c                 ó    •— t         ‰| �  «       S )a  
        Enable loss scaling or not.

        Returns:
            bool: enable loss scaling return True else return False.

        Examples:
            .. code-block:: python

                >>> # doctest: +REQUIRES(env:GPU, env:XPU)
                >>> import paddle
                >>> scaler = paddle.amp.GradScaler(
                ...     enable=True,
                ...     init_loss_scaling=1024,
                ...     incr_ratio=2.0,
                ...     decr_ratio=0.5,
                ...     incr_every_n_steps=1000,
                ...     decr_every_n_nan_or_inf=2,
                ...     use_dynamic_loss_scaling=True
                ... )
                >>> enable = scaler.is_enable()
                >>> print(enable)
                True
        )r®   r   ©r?   r¯   s    €r   r   zGradScaler.is_enableM  s   ø€ ô2 ‰wÑ Ó"Ð"r   c                 ó    •— t         ‰| �  «       S )av  
        Whether to use dynamic loss scaling.

        Returns:
            bool: if fixed loss_scaling is used return False, if the loss scaling is updated dynamicly return true.

        Examples:
            .. code-block:: python

                >>> # doctest: +REQUIRES(env:GPU, env:XPU)
                >>> import paddle
                >>> scaler = paddle.amp.GradScaler(
                ...     enable=True,
                ...     init_loss_scaling=1024,
                ...     incr_ratio=2.0,
                ...     decr_ratio=0.5,
                ...     incr_every_n_steps=1000,
                ...     decr_every_n_nan_or_inf=2,
                ...     use_dynamic_loss_scaling=True
                ... )
                >>> use_dynamic_loss_scaling = scaler.is_use_dynamic_loss_scaling()
                >>> print(use_dynamic_loss_scaling)
                True
        )r®   r�   r¹   s    €r   r�   z&GradScaler.is_use_dynamic_loss_scalingh  ó   ø€ ô2 ‰wÑ2Ó4Ð4r   c                 ó    •— t         ‰| �  «       S )a%  
        Return the initial loss scaling factor.

        Reurns:
            float:  the initial loss scaling factor.

        Examples:
            .. code-block:: python

                >>> # doctest: +REQUIRES(env:GPU, env:XPU)
                >>> import paddle
                >>> scaler = paddle.amp.GradScaler(
                ...     enable=True,
                ...     init_loss_scaling=1024,
                ...     incr_ratio=2.0,
                ...     decr_ratio=0.5,
                ...     incr_every_n_steps=1000,
                ...     decr_every_n_nan_or_inf=2,
                ...     use_dynamic_loss_scaling=True
                ... )
                >>> init_loss_scaling = scaler.get_init_loss_scaling()
                >>> print(init_loss_scaling)
                1024
        )r®   rƒ   r¹   s    €r   rƒ   z GradScaler.get_init_loss_scalingƒ  s   ø€ ô2 ‰wÑ,Ó.Ð.r   c                 ó$   •— t         ‰| �  |«       y)a  
        Set the initial loss scaling factor by `new_init_loss_scaling`.

        Args:
            new_init_loss_scaling(float):  The new_init_loss_scaling used to update initial loss scaling factor.

        Examples:
            .. code-block:: python

                >>> # doctest: +REQUIRES(env:GPU, env:XPU)
                >>> import paddle
                >>> scaler = paddle.amp.GradScaler(
                ...     enable=True,
                ...     init_loss_scaling=1024,
                ...     incr_ratio=2.0,
                ...     decr_ratio=0.5,
                ...     incr_every_n_steps=1000,
                ...     decr_every_n_nan_or_inf=2,
                ...     use_dynamic_loss_scaling=True
                ... )
                >>> print(scaler.get_init_loss_scaling())
                1024
                >>> new_init_loss_scaling = 1000
                >>> scaler.set_init_loss_scaling(new_init_loss_scaling)
                >>> print(scaler.get_init_loss_scaling())
                1000
        N)r®   r†   )r?   r…   r¯   s     €r   r†   z GradScaler.set_init_loss_scalingž  s   ø€ ô8 	‰Ñ%Ð&;Õ<r   c                 ó    •— t         ‰| �  «       S )a=  
        Return the multiplier to use when increasing the loss scaling.

        Reurns:
            float:  the multiplier to use when increasing the loss scaling.

        Examples:
            .. code-block:: python

                >>> # doctest: +REQUIRES(env:GPU, env:XPU)
                >>> import paddle
                >>> scaler = paddle.amp.GradScaler(
                ...     enable=True,
                ...     init_loss_scaling=1024,
                ...     incr_ratio=2.0,
                ...     decr_ratio=0.5,
                ...     incr_every_n_steps=1000,
                ...     decr_every_n_nan_or_inf=2,
                ...     use_dynamic_loss_scaling=True
                ... )
                >>> incr_ratio = scaler.get_incr_ratio()
                >>> print(incr_ratio)
                2.0
        )r®   r‰   r¹   s    €r   r‰   zGradScaler.get_incr_ratio¼  ó   ø€ ô2 ‰wÑ%Ó'Ð'r   c                 ó$   •— t         ‰| �  |«       y)a  
        Set the multiplier to use when increasing the loss scaling by `new_incr_ratio`, `new_incr_ratio` should > 1.0.

        Args:
            new_incr_ratio(float):  The new_incr_ratio used to update the multiplier to use when increasing the loss scaling.

        Examples:
            .. code-block:: python

                >>> # doctest: +REQUIRES(env:GPU, env:XPU)
                >>> import paddle
                >>> scaler = paddle.amp.GradScaler(
                ...     enable=True,
                ...     init_loss_scaling=1024,
                ...     incr_ratio=2.0,
                ...     decr_ratio=0.5,
                ...     incr_every_n_steps=1000,
                ...     decr_every_n_nan_or_inf=2,
                ...     use_dynamic_loss_scaling=True
                ... )
                >>> print(scaler.get_incr_ratio())
                2.0
                >>> new_incr_ratio = 3.0
                >>> scaler.set_incr_ratio(new_incr_ratio)
                >>> print(scaler.get_incr_ratio())
                3.0
        N)r®   r�   )r?   rŒ   r¯   s     €r   r�   zGradScaler.set_incr_ratio×  ó   ø€ ô8 	‰Ñ˜~Õ.r   c                 ó    •— t         ‰| �  «       S )aV  
        Get the less-than-one-multiplier to use when decreasing the loss scaling.

        Reurns:
            float:  the less-than-one-multiplier to use when decreasing the loss scaling.

        Examples:
            .. code-block:: python

                >>> # doctest: +REQUIRES(env:GPU, env:XPU)
                >>> import paddle
                >>> scaler = paddle.amp.GradScaler(
                ...     enable=True,
                ...     init_loss_scaling=1024,
                ...     incr_ratio=2.0,
                ...     decr_ratio=0.5,
                ...     incr_every_n_steps=1000,
                ...     decr_every_n_nan_or_inf=2,
                ...     use_dynamic_loss_scaling=True
                ... )
                >>> decr_ratio = scaler.get_decr_ratio()
                >>> print(decr_ratio)
                0.5
        )r®   r‘   r¹   s    €r   r‘   zGradScaler.get_decr_ratioõ  r¿   r   c                 ó$   •— t         ‰| �  |«       y)a7  
        Set the less-than-one-multiplier to use when decreasing the loss scaling by `new_incr_ratio`, `new_decr_ratio` should < 1.0.

        Args:
            new_decr_ratio(float):  The new_decr_ratio used to update the less-than-one-multiplier to use when decreasing the loss scaling.

        Examples:
            .. code-block:: python

                >>> # doctest: +REQUIRES(env:GPU, env:XPU)
                >>> import paddle
                >>> scaler = paddle.amp.GradScaler(
                ...     enable=True,
                ...     init_loss_scaling=1024,
                ...     incr_ratio=2.0,
                ...     decr_ratio=0.5,
                ...     incr_every_n_steps=1000,
                ...     decr_every_n_nan_or_inf=2,
                ...     use_dynamic_loss_scaling=True
                ... )
                >>> print(scaler.get_decr_ratio())
                0.5
                >>> new_decr_ratio = 0.1
                >>> scaler.set_decr_ratio(new_decr_ratio)
                >>> print(scaler.get_decr_ratio())
                0.1
        N)r®   r”   )r?   r“   r¯   s     €r   r”   zGradScaler.set_decr_ratio  rÁ   r   c                 ó    •— t         ‰| �  «       S )a®  
        Return the num `n`, `n` represent increases loss scaling every `n` consecutive steps with finite gradients.

        Reurns:
            int:  the num `n`, `n` represent increases loss scaling every `n` consecutive steps with finite gradients.

        Examples:
            .. code-block:: python

                >>> # doctest: +REQUIRES(env:GPU, env:XPU)
                >>> import paddle
                >>> scaler = paddle.amp.GradScaler(
                ...     enable=True,
                ...     init_loss_scaling=1024,
                ...     incr_ratio=2.0,
                ...     decr_ratio=0.5,
                ...     incr_every_n_steps=1000,
                ...     decr_every_n_nan_or_inf=2,
                ...     use_dynamic_loss_scaling=True
                ... )
                >>> incr_every_n_steps = scaler.get_incr_every_n_steps()
                >>> print(incr_every_n_steps)
                1000
        )r®   r—   r¹   s    €r   r—   z!GradScaler.get_incr_every_n_steps.  s   ø€ ô2 ‰wÑ-Ó/Ð/r   c                 ó$   •— t         ‰| �  |«       y)a—  
        Set the num `n` by `new_incr_every_n_steps`, `n` represent increases loss scaling every `n` consecutive steps with finite gradients.

        Args:
            new_incr_every_n_steps(int):  The new_incr_every_n_steps used to update the num `n`, `n` represent increases loss scaling every `n` consecutive steps with finite gradients.

        Examples:
            .. code-block:: python

                >>> # doctest: +REQUIRES(env:GPU, env:XPU)
                >>> import paddle
                >>> scaler = paddle.amp.GradScaler(
                ...     enable=True,
                ...     init_loss_scaling=1024,
                ...     incr_ratio=2.0,
                ...     decr_ratio=0.5,
                ...     incr_every_n_steps=1000,
                ...     decr_every_n_nan_or_inf=2,
                ...     use_dynamic_loss_scaling=True
                ... )
                >>> print(scaler.get_incr_every_n_steps())
                1000
                >>> new_incr_every_n_steps = 2000
                >>> scaler.set_incr_every_n_steps(new_incr_every_n_steps)
                >>> print(scaler.get_incr_every_n_steps())
                2000
        N)r®   rš   )r?   r™   r¯   s     €r   rš   z!GradScaler.set_incr_every_n_stepsI  s   ø€ ô8 	‰Ñ&Ð'=Õ>r   c                 ó    •— t         ‰| �  «       S )aÂ  
        Return the num `n`, `n` represent decreases loss scaling every `n` accumulated steps with nan or inf gradients.

        Reurns:
            int:  the num `n`, `n` represent decreases loss scaling every `n` accumulated steps with nan or inf gradients.

        Examples:
            .. code-block:: python

                >>> # doctest: +REQUIRES(env:GPU, env:XPU)
                >>> import paddle
                >>> scaler = paddle.amp.GradScaler(
                ...     enable=True,
                ...     init_loss_scaling=1024,
                ...     incr_ratio=2.0,
                ...     decr_ratio=0.5,
                ...     incr_every_n_steps=1000,
                ...     decr_every_n_nan_or_inf=2,
                ...     use_dynamic_loss_scaling=True
                ... )
                >>> decr_every_n_nan_or_inf = scaler.get_decr_every_n_nan_or_inf()
                >>> print(decr_every_n_nan_or_inf)
                2
        )r®   r�   r¹   s    €r   r�   z&GradScaler.get_decr_every_n_nan_or_infg  r»   r   c                 ó$   •— t         ‰| �  |«       y)a¾  
        Set the num `n` by `new_decr_every_n_nan_or_inf`, `n` represent decreases loss scaling every `n` accumulated steps with nan or inf gradients.

        Args:
            new_decr_every_n_nan_or_inf(int):  The new_decr_every_n_nan_or_inf used to update the num `n`, `n` represent decreases loss scaling every `n` accumulated steps with nan or inf gradients.

        Examples:
            .. code-block:: python

                >>> # doctest: +REQUIRES(env:GPU, env:XPU)
                >>> import paddle
                >>> scaler = paddle.amp.GradScaler(
                ...     enable=True,
                ...     init_loss_scaling=1024,
                ...     incr_ratio=2.0,
                ...     decr_ratio=0.5,
                ...     incr_every_n_steps=1000,
                ...     decr_every_n_nan_or_inf=2,
                ...     use_dynamic_loss_scaling=True
                ... )
                >>> print(scaler.get_decr_every_n_nan_or_inf())
                2
                >>> new_decr_every_n_nan_or_inf = 3
                >>> scaler.set_decr_every_n_nan_or_inf(new_decr_every_n_nan_or_inf)
                >>> print(scaler.get_decr_every_n_nan_or_inf())
                3
        N)r®   r    )r?   rŸ   r¯   s     €r   r    z&GradScaler.set_decr_every_n_nan_or_inf‚  s   ø€ ô8 	‰Ñ+Ð,GÕHr   c                 ó    •— t         ‰| �  «       S )a.  
        Returns the state of the scaler as a `dict`, If this instance is not enabled, returns an empty dict.

        Returns:
            A dict of scaler includes:
            scale (tensor): The loss scaling factor.
            incr_ratio(float): The multiplier to use when increasing the loss scaling.
            decr_ratio(float): The less-than-one-multiplier to use when decreasing the loss scaling.
            incr_every_n_steps(int): Increases loss scaling every n consecutive steps with finite gradients.
            decr_every_n_nan_or_inf(int): Decreases loss scaling every n accumulated steps with nan or inf gradients.
            incr_count(int): The number of recent consecutive unskipped steps.
            decr_count(int): The number of recent consecutive skipped steps.
            use_dynamic_loss_scaling(bool): Whether to use dynamic loss scaling. If False, fixed loss_scaling is used. If True, the loss scaling is updated dynamicly. Default is True.


        Examples:

            .. code-block:: python

                >>> # doctest: +REQUIRES(env:GPU, env:XPU)
                >>> import paddle

                >>> scaler = paddle.amp.GradScaler(
                ...     enable=True,
                ...     init_loss_scaling=1024,
                ...     incr_ratio=2.0,
                ...     decr_ratio=0.5,
                ...     incr_every_n_steps=1000,
                ...     decr_every_n_nan_or_inf=2,
                ...     use_dynamic_loss_scaling=True
                ... )
                >>> scaler_state = scaler.state_dict()
        )r®   r¥   r¹   s    €r   r¥   zGradScaler.state_dict   s   ø€ ôD ‰wÑ!Ó#Ð#r   c                 ó$   •— t         ‰| �  |«       y)a;  
        Loads the scaler state.

        Args:
            state_dict(dict): scaler state.  Should be an object returned from a call to `GradScaler.state_dict()`.

        Examples:

            .. code-block:: python

                >>> # doctest: +REQUIRES(env:GPU, env:XPU)
                >>> import paddle

                >>> scaler = paddle.amp.GradScaler(
                ...     enable=True,
                ...     init_loss_scaling=1024,
                ...     incr_ratio=2.0,
                ...     decr_ratio=0.5,
                ...     incr_every_n_steps=1000,
                ...     decr_every_n_nan_or_inf=2,
                ...     use_dynamic_loss_scaling=True
                ... )
                >>> scaler_state = scaler.state_dict()
                >>> scaler.load_state_dict(scaler_state)
        N)r®   r§   )r?   r¥   r¯   s     €r   r§   zGradScaler.load_state_dictÄ  s   ø€ ô4 	‰Ñ 
Õ+r   )Tg      ð@r¨   r©   iÐ  r   T)r   r   r   rª   rH   rO   rS   r³   rµ   r·   r   r�   rƒ   r†   r‰   r�   r‘   r”   r—   rš   r�   r    r¥   r§   Ú__classcell__)r¯   s   @r   r¬   r¬   B  s‘   ø„ ñ1ðj Ø!ØØØØ !Ø!%õ
ô("ô@"<òH:Kòxô@!+ôF#ô65ô6/ô6=ô<(ô6/ô<(ô6/ô<0ô6?ô<5ô6Iô<"$÷H,ð ,r   r¬   )r'   Úcollectionsr   Úenumr   r¤   r2   Úpaddler   r   Úpaddle.baser   Úpaddle.base.data_feederr   Úpaddle.base.dygraphr	   Úpaddle.base.frameworkr
   r   Úpaddle.frameworkr   Ú	auto_castr   r   r   r   r¬   r   r   r   Ú<module>rÔ      sV   ðó Ý #Ý ã ç (Ý Ý .Ý +ß ?Ý ,å 'ô�Tô ò*÷VPñ VPôr\
,�õ \
,r   