Ë
    •\;j_? ã                   ól  — d dl Z d dlZd dlmZ d dlZd dlZd dlmZ	 d dlm
Z
 d dlmZmZ d dlmZ d dlmZmZmZmZmZmZmZmZmZ d dlmZ dd	lmZmZ dd
lm Z m!Z! ddl"m#Z# ddl$m%Z% ddl&m'Z' g Z( e)ejT                  jW                  dd «      «      Z,ejZ                  	 	 	 	 	 dd„«       Z. G d„ d«      Z/y)é    N)Údefaultdict)Ú_C_ops)Ú	parameterÚset_parameter)Úcore)	ÚVariableÚ_current_expected_placeÚdefault_main_programÚdevice_guardÚin_dygraph_modeÚin_dynamic_or_pir_modeÚin_pir_modeÚ
name_scopeÚuse_pir_api)ÚL2Decayé   )Ú	frameworkÚunique_name)Ú_get_no_grad_set_nameÚappend_backward)Ú	Parameter)ÚLayerHelperé   ©ÚLRSchedulerÚ$FLAGS_shard_bypass_dygraph_optimizerc                 óÌ  — ddl m}m} t        «       }|j                  dk(  sJ d«       ‚|j                  «       }	| D ]  }
|
j                  |	k(  rŒJ d«       ‚  ||	«        ||	«      }|€|j                  «       j                  «       }|j                  || «      \  }}|j                  ||«      \  }}g }|D ]B  }|€Œ|	j                  j                  |j                  «      }|dk\  sJ ‚|j                  |«       ŒD |j                  t!        |«      «       |j#                  |«       t%        |«      dk(  r||fg}|S g }t'        |«      D ]  \  }}|j                  |||   f«       Œ |S )Nr   )Ú	TransformÚ	orig2primr   zHThe append_backward_new interface is designed to process only one block.z@variable in loss_list should be in current block of main program)Úpaddle.incubate.autograd.primxr   r   r
   Ú
num_blocksÚcurrent_blockÚblockÚglobal_blockÚall_parametersÚ	linearizeÚ	transposeÚopsÚindexÚopÚappendÚ	erase_opsÚsortedÚ
erase_dotsÚlenÚ	enumerate)Ú	loss_listÚparameter_listÚno_grad_setÚ	callbacksÚcheckpointsÚdistop_contextr   r   Úprogramr#   ÚelÚadÚ	param_dotÚloss_dotÚloss_barÚ	param_barÚ
op_indexesÚvarÚop_indexÚparams_and_gradsÚiÚparams                         úcG:\00. PROJECTS\API\Inventory\templateJSON\kerjaOCR\Lib\site-packages\paddle/optimizer/optimizer.pyÚappend_backward_newrE   4   s�  € ÷ Dä"Ó$€Gà×Ñ˜aÒðRàQóRØà×!Ñ!Ó#€EÛˆà�H‰H˜Óð	NàMó	NØð ñ
 ˆeÔÙ	�5Ó	€BØÐØ ×-Ñ-Ó/×>Ñ>Ó@ˆØŸ,™, ~°yÓAÑ€IˆxØŸ,™, x°Ó;Ñ€Hˆið €JÛˆØ‰?Ø—y‘y—‘ s§v¡vÓ.ˆHØ˜q’=Ð �=Ø×Ñ˜hÕ'ð	 ð ‡L�L”˜
Ó#Ô$Ø‡M�M�)Ôä
ˆ>Ó˜aÒØ+¨YÐ7Ð8Ðð
 Ðð ÐÜ! .Ö1‰HˆAˆuØ×#Ñ# U¨I°a©LÐ$9Õ:ð 2àÐó    c                   ó°  — e Zd ZdZ ej
                  «       	 	 	 	 d-d„«       Zd„ Zd„ Zd„ Z	d„ Z
ej                  d„ «       Zej                  d	„ «       Zd
„ Zd„ Zej                  d„ «       Zej                  d„ «       Zd„ Zd.d„Zd„ Zd„ Zd„ Zd„ Zd„ Zd„ Zd„ Z	 	 	 	 	 d/d„Zd„ Zd„ Zd„ Zd„ Z 	 d0d„Z!	 d0d„Z"	 	 	 	 d-d„Z#d„ Z$	 d0d „Z%d.d!„Z&	 d.d"„Z'd.d#„Z(ejR                  d1d$„«       Z* ej
                  «       	 d2d%„«       Z+d&„ Z, ej
                  «       ejR                  d'„ «       «       Z-d(„ Z.d)„ Z/ej                  d*„ «       Z0ej                  d+„ «       Z1d,„ Z2y)3Ú	OptimizeraW  Optimizer Base class.

    Define the common interface of an optimizer.
    User should not use this class directly,
    but need to use one of it's implementation.

    Args:
        learning_rate (float|LRScheduler): The learning rate used to update ``Parameter``.
            It can be a float value or any subclass of ``LRScheduler`` .
        parameters (list|tuple, optional): List/Tuple of ``Tensor`` names to update to minimize ``loss``. \
            This parameter is required in dygraph mode. And you can specify different options for \
            different parameter groups such as the learning rate, weight decay, etc, \
            then the parameters are list of dict. Note that the learning_rate in paramter groups \
            represents the scale of base learning_rate. \
            The default value is None in static graph mode, at this time all parameters will be updated.
        weight_decay (float|WeightDecayRegularizer, optional): The strategy of regularization. \
            It canbe a float value as coeff of L2 regularization or \
            :ref:`api_paddle_regularizer_L1Decay`, :ref:`api_paddle_regularizer_L2Decay`.
            If a parameter has set regularizer using :ref:`api_paddle_ParamAttr` already, \
            the regularization setting here in optimizer will be ignored for this parameter. \
            Otherwise, the regularization setting here in optimizer will take effect. \
            Default None, meaning there is no regularization.
        grad_clip (GradientClipBase, optional): Gradient cliping strategy, it's an instance of \
            some derived class of ``GradientClipBase`` . There are three cliping strategies \
            ( :ref:`api_paddle_nn_ClipGradByGlobalNorm` , :ref:`api_paddle_nn_ClipGradByNorm` , \
            :ref:`api_paddle_nn_ClipGradByValue` ). Default None, meaning there is no gradient clipping.
        name (str, optional): Normally there is no need for user to set this property.
            For more information, please refer to :ref:`api_guide_Name`.
            The default value is None.

    Returns:
       Base class for optimizer.

    Examples:
        .. code-block:: python

            >>> # Take the subclass adam as an example
            >>> import paddle
            >>> linear = paddle.nn.Linear(10, 10)
            >>> inp = paddle.uniform(shape=[10, 10], min=-0.1, max=0.1)
            >>> out = linear(inp)
            >>> loss = paddle.mean(out)
            >>> adam = paddle.optimizer.Adam(learning_rate=0.1,
            ...         parameters=linear.parameters())
            >>> loss.backward()
            >>> adam.step()
            >>> adam.clear_grad()

            >>> #Take the subclass sgd as an example
            >>> #optimize parameters in linear_1 and linear2 in different options.
            >>> #Note that the learning_rate of linear_2 is 0.01.
            >>> linear_1 = paddle.nn.Linear(10, 10)
            >>> linear_2 = paddle.nn.Linear(10, 10)
            >>> inp = paddle.uniform(shape=[10, 10], min=-0.1, max=0.1)
            >>> out = linear_1(inp)
            >>> out = linear_2(out)
            >>> loss = paddle.mean(out)
            >>> sgd = paddle.optimizer.SGD(
            ...     learning_rate=0.1,
            ...     parameters=[{
            ...         'params': linear_1.parameters()
            ...     }, {
            ...         'params': linear_2.parameters(),
            ...         'weight_decay': 0.001,
            ...         'learning_rate': 0.1
            ...     }],
            ...     weight_decay=0.01)
            >>> loss.backward()
            >>> sgd.step()
            >>> sgd.clear_grad()

    Nc                 óø  — |�ƒt        |t        j                  t        j                  j                  f«      r#t        dj                  t        |«      «      «      ‚t        |t        «      rt        d«      ‚t        |«      | _
        nd | _
        || _        t        j                  «       rˆ| j                  €t        d«      ‚|�ot        | j                  d   t        «      sR| j                  D ]C  }t        |d«      sŒ|j                   €Œt#        j$                  d|j'                  «       z  «        n t        |t(        t*        f«      st        dt        |«      z  «      ‚|�9t        |t        j,                  j.                  j0                  «      st        d«      ‚t        |t(        «      rt3        |«      | _        n|| _        || _        || _        d | _        | j                  r|t        | j                  d   t        «      rA| j                  D ]  }d	|v rŒJ d
«       ‚ | j                  d   d	   d   j<                  | _        n| j                  d   j<                  | _        i | _        tA        d„ «      | _!        d | _"        g | _#        i | _$        i | _%        | jL                  | _'        | j4                  | j6                  dœ| _(        g | _)        | j                  rNt        | j                  d   t        «      r1| j                  D ]!  }| jU                  |jW                  «       «       Œ# n| j                  | _)        d | _,        | j[                  «       | _.        i | _/        ta        «       | _1        i | _2        | jg                  «        y )Nzt`parameters` argument given to the optimizer should be an iterable of paddle Tensors, but got argument type is `{}`.zv`parameters` argument should not get dict type, if parameter groups is needed, please set `parameters` as list of dictzNparameters argument given to the Optimizer should not be None in dygraph mode.r   ÚregularizerzÒIf regularizer of a Parameter has been set by 'paddle.ParamAttr' or 'static.WeightNormParamAttr' already. The weight_decay[%s] in Optimizer will not take effect, and it will only be applied to other Parameters!z9learning rate should be float or LRScheduler, got %s herezE'grad_clip' should be an instance of GradientClipBase's derived classÚparamszYparams should be set in parameters if parameter groups are optimized in different optionsc                  ó   — i S ©N© rN   rF   rD   Ú<lambda>z$Optimizer.__init__.<locals>.<lambda>  s   € ±rF   )Úweight_decayÚ	grad_clip)4Ú
isinstanceÚpaddleÚTensorr   ÚeagerÚ	TypeErrorÚformatÚtypeÚdictÚlistÚ_parameter_listÚ_namer   r   ÚAttributeErrorÚhasattrrJ   ÚloggingÚinfoÚ__str__Úfloatr   ÚnnÚclipÚGradientClipBaser   ÚregularizationÚ
_grad_clipÚ_learning_rateÚ_dtypeÚdtypeÚ_learning_rate_mapr   Ú_accumulatorsÚhelperÚ_opti_name_listÚ_accumulators_holderÚ_param_device_mapÚ
clear_gradÚclear_gradientsÚ_default_dictÚ_param_groupsÚ_add_param_groupÚcopyÚ_use_multi_tensorÚ_create_multi_tensor_dictÚ_param_dictÚ_auxiliary_varsÚsetÚ_already_create_accumulaterÚ_master_weightsÚ_create_master_grad_states)ÚselfÚlearning_rateÚ
parametersrP   rQ   ÚnamerC   Úparam_groups           rD   Ú__init__zOptimizer.__init__®   s5  € ð Ð!ô ˜*¤v§}¡}´d·j±j×6GÑ6GÐ&HÔIÜðTßTZÑTZÜ˜ZÓ(óUóð ô ˜*¤dÔ+Üð'óð ô
 $(¨
Ó#3ˆDÕ à#'ˆDÔ àˆŒ
Ü×$Ñ$Ô&Ø×#Ñ#Ð+Ü$Ødóð ð Ð'Ü! $×"6Ñ"6°qÑ"9¼4Ô@Ø!%×!5Ô!5˜ä# E¨=Õ9Ø %× 1Ñ 1Ñ =ä#ŸL™Lð!Kà".×"6Ñ"6Ó"8ñ!9ôñ
 "ð "6ô ˜-¬%´Ð)=Ô>ÜØKÜ�}Ó%ñ&óð ð Ð Ü˜i¬¯©¯©×)HÑ)HÔIÜØ[óð ô �l¤EÔ*Ü")¨,Ó"7ˆDÕà".ˆDÔØ#ˆŒØ+ˆÔàˆŒà×ÒÜ˜$×.Ñ.¨qÑ1´4Ô8Ø#'×#7Ô#7�Kà  KÒ/ðsàrósØ/ð $8ð #×2Ñ2°1Ñ5°hÑ?ÀÑB×HÑH�•à"×2Ñ2°1Ñ5×;Ñ;�”ð #%ˆÔô
 )©Ó4ˆÔØˆŒØ!ˆÔØ$&ˆÔ!Ø!#ˆÔØ#Ÿ™ˆÔà ×/Ñ/ØŸ™ñ
ˆÔð
  ˆÔØ×Ò¤J¨t×/CÑ/CÀAÑ/FÌÔ$MØ#×3Ô3�Ø×%Ñ% k×&6Ñ&6Ó&8Õ9ñ  4ð "&×!5Ñ!5ˆDÔð "&ˆÔà×9Ñ9Ó;ˆÔØ!ˆÔÜ+.«5ˆÔ(à!ˆÔà×'Ñ'Õ)rF   c                 ó    — i | _         d| _        y )NF)Ú_master_gradsÚ_master_grad©r   s    rD   r~   z$Optimizer._create_master_grad_states"  s   € àˆÔØ!ˆÕrF   c                 ó"   — || j                   |<   y rM   )rz   )r   ÚkeyÚvals      rD   Ú_set_auxiliary_varzOptimizer._set_auxiliary_var'  s   € Ø$'ˆ×Ñ˜SÒ!rF   c                 óÂ   — | j                   �t        | j                   «      nd}t        |«      D �cg c]  }g ‘Œ c}t        |«      D �cg c]  }g ‘Œ c}dœS c c}w c c}w )Nr   )ÚFP32_LODTensorÚFP16_LODTensor)rt   r/   Úrange)r   ÚnÚ_s      rD   rx   z#Optimizer._create_multi_tensor_dict*  s[   € Ø'+×'9Ñ'9Ð'EŒC�×"Ñ"Ô#È1ˆä+0°¬8Ó4©8 ašr¨8Ñ4Ü+0°¬8Ó4©8 ašr¨8Ñ4ñ
ð 	
ùÚ4ùÚ4s   ±	AÁ		Ac                 ó:   — | j                   j                  |d «      S rM   )rz   Úget)r   rŠ   s     rD   Ú_get_auxiliary_varzOptimizer._get_auxiliary_var1  s   € Ø×#Ñ#×'Ñ'¨¨TÓ2Ð2rF   c                 óö  — i }| j                   j                  «       D ]o  \  }}|j                  «       D ]W  \  }}|||j                  <   t        j                  «       sŒ*|j                  «       j                  «       ||j                  dz   <   ŒY Œq t        | d«      r't        | j                  «      dk7  r| j                  |d<   t        | j                  t        «      r| j                  j                  «       |d<   |S )aÛ  
        Get state dict information from optimizer. It contain all the tensor used by optimizer. For Adam optimizer, contains beta1, beta2, momentum etc. If LRScheduler have been used, global_step will be include in state dict.
        If the optimizer never be called(minimize function), the state_dict is empty.

        Args:
            None

        Returns:
            state_dict(dict) : dict contains all the Tensor used by optimizer

        Examples:
            .. code-block:: python

                >>> import paddle
                >>> emb = paddle.nn.Embedding(10, 10)

                >>> adam = paddle.optimizer.Adam(0.001, parameters=emb.parameters())
                >>> state_dict = adam.state_dict()

        ú.SCALE_VALUEr}   r   Úmaster_weightsÚLR_Scheduler)rl   Úitemsr‚   r   Úis_compiled_with_xpuÚ
get_tensorÚget_xpu_scale_valuer^   r/   r}   rR   rh   r   Ú
state_dict)r   rž   ÚkÚvÚ	para_nameÚvar_tmps         rD   rž   zOptimizer.state_dict4  sâ   € ð, ˆ
Ø×&Ñ&×,Ñ,Ö.‰DˆAˆqØ&'§g¡g¦iÑ"�	˜7Ø+2�
˜7Ÿ<™<Ñ(ä×,Ñ,Õ.ð  ×*Ñ*Ó,×@Ñ@ÓBð ØŸ™ ~Ñ5òñ	 '0ð /ô �4Ð*Ô+Ü�4×'Ñ'Ó(¨AÒ-Ø/3×/CÑ/C�
Ð+Ñ,ä�d×)Ñ)¬;Ô7Ø)-×)<Ñ)<×)GÑ)GÓ)IˆJ�~Ñ&ØÐrF   c           
      ó°  — t        | j                  t        «      r| j                  j                  |d   «       |j	                  «       }d|v r|j                  d«       d|v r't        | d«      r
|d   | _        |j                  d«       || _        | j                  j                  «       D �])  \  }}|j                  «       D �]  \  }}|j                  |v sJ d|j                  › d�«       ‚|j                  «       }|j                  «       }t        j                  «       r.|j!                  |j#                  |j                  dz   d«      «       t%        j&                  |«      }||j                     }	t        |	t(        «      rt%        j&                  |	«      }
nxt        |	t        j*                  j,                  «      rt%        j&                  |	«      }
n>t        |	t$        j.                  «      r|	}
n!t1        dt3        t5        |	«      «      › d	�«      ‚|j6                  |
j6                  k(  s6J d
j9                  |j                  |j6                  |
j6                  «      «       ‚|j:                  |
j:                  k(  s6J dj9                  |j                  |j:                  |
j:                  «      «       ‚|j=                  |
t?        j@                  «       «       �Œ �Œ, y)a:  
        Load optimizer state dict. For Adam optimizer, contains beta1, beta2, momentum etc. If LRScheduler have been used, global_step will be changed.

        Args:
            state_dict(dict) : Dict contains all the Tensor needed by optimizer
        Return:
            None

        Examples:
            .. code-block:: python

                >>> import paddle

                >>> emb = paddle.nn.Embedding(10, 10)

                >>> layer_state_dict = emb.state_dict()
                >>> paddle.save(layer_state_dict, "emb.pdparams")

                >>> scheduler = paddle.optimizer.lr.NoamDecay(
                ...     d_model=0.01, warmup_steps=100, verbose=True)
                >>> adam = paddle.optimizer.Adam(
                ...     learning_rate=scheduler,
                ...     parameters=emb.parameters())
                >>> opt_state_dict = adam.state_dict()
                >>> paddle.save(opt_state_dict, "adam.pdopt")

                >>> opti_state_dict = paddle.load("adam.pdopt")
                >>> adam.set_state_dict(opti_state_dict)

        r™   r˜   r}   zoptimizer Tensor z
 not foundr—   ç      ð¿zState dict type z not supprtzkParameter shape not match, Dygraph Parameter [ {} ] need tensor with shape {} but load tensor with shape {}zlParameter dtype not match, Dygraph Parameter [ {} ] need tensor with dtype {}  but load tensor with dtype {}N)!rR   rh   r   Úset_state_dictrv   Úpopr^   r}   ro   rl   rš   r‚   Úvaluerœ   r   r›   Úset_xpu_scale_valuer”   ÚnpÚarrayr   rU   rT   ÚndarrayÚRuntimeErrorÚstrrX   ÚshaperW   rj   r{   r   r	   )r   rž   rŸ   r    r¡   r¢   r?   ÚtensorÚmodel_npÚ	load_paraÚload_para_nps              rD   r¥   zOptimizer.set_state_dict\  so  € ô@ �d×)Ñ)¬;Ô7Ø×Ñ×.Ñ.¨z¸.Ñ/IÔJð  —_‘_Ó&ˆ
Ø˜ZÑ'Ø�N‰N˜>Ô*Ø˜zÑ)Ü�tÐ.Ô/Ø'1Ð2BÑ'C�Ô$Ø�N‰NÐ+Ô,Ø$.ˆÔ!Ø×&Ñ&×,Ñ,×.‰DˆAˆqØ&'§g¡g§iÑ"�	˜7à—L‘L JÑ.ð@à& w§|¡| n°JÐ?ó@Ø.ð —m‘m“o�ØŸ™Ó)�ä×,Ñ,Ô.Ø×.Ñ.Ø"Ÿ™ w§|¡|°nÑ'DÀdÓKôô Ÿ8™8 FÓ+�à& w§|¡|Ñ4�	ä˜i¬Ô2Ü#%§8¡8¨IÓ#6‘LÜ 	¬4¯:©:×+<Ñ+<Ô=Ü#%§8¡8¨IÓ#6‘LÜ 	¬2¯:©:Ô6Ø#,‘Lä&Ø*¬3¬t°I«Ó+?Ð*@ÀÐLóð ð
 —N‘N l×&8Ñ&8Ò8ðð A÷  Hñ  HØ—M‘M 8§>¡>°<×3EÑ3EóóØ8ð —N‘N l×&8Ñ&8Ò8ðð B÷  Iñ  IØ—M‘M 8§>¡>°<×3EÑ3EóóØ8ð
 —
‘
˜<¬×)JÑ)JÓ)LÖMòQ '0ñ /rF   c                 ó   — | j                   S rM   )rn   rˆ   s    rD   Úget_opti_var_name_listz Optimizer.get_opti_var_name_list´  s   € Ø×#Ñ#Ð#rF   c                 ó˜   ‡ — ˆ fd„}t         j                  j                  j                  «       5   |«        d d d «       y # 1 sw Y   y xY w)Nc                  ó6  •— ‰j                   €t        j                  «       n‰j                   } t        j                  «       dk7  r| t        j                  k(  s*t        j                  «       dk7  r#| t        j                  k(  rt        j
                  n| } t        ‰j                  t        «      �rê‰j                  «       }t        «       �rÂt        j                  j                  «       }t        j                  j                  «       }t        j                  d«      }t!        ‰j                  «       «      }t        j                  j#                  |«      5  t        j$                  j&                  j)                  |¬«      }t        j*                  j,                  j/                  g | «      } |||j1                  «       «      }d|_        t5        ||«       d d d «       |j7                  |«       t        |t        j*                  j8                  «      �sˆ|‰j                  _        t        j                  j#                  |«      5  t=        || g «      }	d d d «       d	_        d|	_        ‰j                  |_         |	|_!        |	‰jD                  |<   y t        |tF        jH                  «      s“t        j                  d«      }|‰j                  _        ‰jJ                  jM                  |g dd| ¬«      }tG        j                  «       }
‰j                  |
_         ||
_!        |‰jD                  tG        j                  «       <   t!        ‰j                  «       «      }‰jJ                  jO                  |t        j$                  j&                  j)                  |¬«      ¬«       y y t        ‰j                  t         «      �rÌ‰j                  «       }t        «       �r,t        |t        j*                  j8                  «      ry tQ        «       }t        | t        jR                  j,                  jT                  «      s)t        j*                  j,                  jW                  | «      }t        j*                  j,                  jY                  | g t        j                  d«      dt        j$                  j&                  j[                  t!        ‰j                  «      ¬«      ¬	«      ‰jD                  t        j                  j                  «       <   y t        |tF        jH                  «      ry t        j                  j]                  t        j                  d«      g t!        ‰j                  «      | d¬
«      ‰jD                  tG        j                  «       <   y y # 1 sw Y   �Œ±xY w# 1 sw Y   �ŒBxY w)NÚfloat16Úbfloat16r€   ©r§   T)r‚   r®   ÚpersistableÚstop_gradientrj   ©ÚinitializerF)rj   r®   r‚   Ú	trainabler½   ©r‚   r®   r§   rj   rº   )/ri   rS   Úget_default_dtyper·   r¸   Úfloat32rR   rh   r   Ú_global_learning_rater   ÚstaticÚdefault_startup_programr
   r   Úgeneraterb   Úprogram_guardrc   r½   ÚConstantÚpirr   ÚParameterMetar$   rº   r   Úmove_parameters_fromÚOpResultÚ	_var_namer   r»   Úlr_schedulerÚlr_varrk   r   r   rm   Úcreate_global_variableÚset_variable_initializerr	   ÚbaseÚDataTypeÚconvert_np_dtype_to_dtype_Úcreate_parameterÚConstantInitializerÚcreate_global_var)Ú	_lr_dtyperÎ   Ústartup_programÚmain_programÚlr_nameÚlr_valuer½   Úparamete_metaÚinit_resultrC   Ú	main_progÚlrÚplaceÚlr_dtyper   s                 €rD   Ú	do_createz9Optimizer._create_global_learning_rate.<locals>.do_create¸  sx  ø€ ð —;‘;Ð&ô ×(Ñ(Ô*à—[‘[ð ô ×0Ñ0Ó2°iÒ?Ø%¬¯©Ò7ô ×0Ñ0Ó2°jÒ@Ø%¬¯©Ò8ô —’ð ð ô ˜$×-Ñ-¬{Õ;Ø×3Ñ3Ó5�ä•=Ü&,§m¡m×&KÑ&KÓ&M�OÜ#)§=¡=×#EÑ#EÓ#G�Lä)×2Ñ2°?ÓC�Gä$ T×%8Ñ%8Ó%:Ó;�HÜŸ™×4Ñ4°_ÕEÜ&,§i¡i×&;Ñ&;×&DÑ&DØ"*ð 'Eó '˜ô )/¯
©
¯©×(EÑ(EØ 	ó)˜ñ '2Ø)¨?×+GÑ+GÓ+Ió'˜ð 37˜Ô/Ü% k°7Ô;÷ Fð !×5Ñ5°oÔFä% f¬f¯j©j×.AÑ.AÕBØ8?˜×+Ñ+Ô5Ü#Ÿ]™]×8Ñ8¸ÕFÜ$-¨g°yÀ"Ó$E˜E÷ Gà.2˜Ô+Ø,0˜Ô)Ø48×4GÑ4G˜Ô1Ø.3˜Ô+Ø@E˜×/Ñ/°Ò=ô & f¬i×.@Ñ.@ÔAÜ"-×"6Ñ"6°Ó"G˜Ø8?˜×+Ñ+Ô5Ø!%§¡×!CÑ!CØ!(Ø"$Ø(,Ø*.Ø"+ð "Dó "˜ô %.×$BÑ$BÓ$D˜	Ø15×1DÑ1D˜	Ô.Ø+1˜	Ô(ð #ð ×/Ñ/Ü%×:Ñ:Ó<ñô  % T×%8Ñ%8Ó%:Ó;�HØ—K‘K×8Ñ8ØÜ$*§I¡I×$9Ñ$9×$BÑ$BØ"*ð %Có %ð 9õ ð= CôH ˜D×/Ñ/´Õ7à×/Ñ/Ó1�Ü•=Ü! "¤f§j¡j×&9Ñ&9Ô:Øä 7Ó 9˜Ü)¨)´V·[±[×5EÑ5E×5NÑ5NÔOä &§
¡
§¡× JÑ JØ$-ó!"ð %ô #ŸJ™JŸO™O×<Ñ<Ø"+Ø"$Ü!,×!5Ñ!5°oÓ!FØ&+Ü(.¯	©	×(=Ñ(=×(QÑ(QÜ&+¨D×,?Ñ,?Ó&@ð )Ró )ð =ó ð ×/Ñ/Ü"ŸM™M×>Ñ>Ó@òô " "¤i×&8Ñ&8Ô9Øô #ŸM™M×;Ñ;Ü!,×!5Ñ!5°oÓ!FØ"$Ü"'¨×(;Ñ(;Ó"<Ø"+Ø(,ð <ó ð ×/Ñ/Ü%×:Ñ:Ó<òð; 8÷e FÑEú÷  GÑFús   ÅA?VÉ VÖVÖV)rS   rÑ   r   Údygraph_guard_if_declarative)r   râ   s   ` rD   Ú_create_global_learning_ratez&Optimizer._create_global_learning_rate·  s4   ø€ ôv	ôp �[‰[×"Ñ"×?Ñ?ÕAÙŒK÷ B×AÑAús   ¯A Á A	c           	      ó^  — t        |t        t        f«      st        dt	        |«      z  «      ‚t        | j
                  t        «      rt        d«      ‚t        |«      | _        | j                  «       }|�¹t        «       rJt        «       }t        j                  |t        |j                  «      t        |«      |j                  |«       yt!        j"                  «       j%                  «       }|j'                  dd|gi|j                  t        |j                  «      t        |«      dœd¬«       yy)	a  
        :api_attr: imperative

        Set the value of the learning rate manually in the optimizer. If the optimizer use LRScheduler,
        this API cannot be invoked, because it will lead to conflict.

        Args:
            value (float): the value of learning rate

        Returns:
            None

        Examples:
            .. code-block:: python

                >>> import paddle
                >>> linear = paddle.nn.Linear(10, 10)

                >>> adam = paddle.optimizer.Adam(0.1, parameters=linear.parameters())

                >>> # set learning rate manually by python float value
                >>> lr_list = [0.2, 0.3, 0.4, 0.5, 0.6]
                >>> for i in range(5):
                ...     adam.set_lr(lr_list[i])
                ...     lr = adam.get_lr()
                ...     print("current lr is {}".format(lr))
                current lr is 0.2
                current lr is 0.3
                current lr is 0.4
                current lr is 0.5
                current lr is 0.6

        zGThe type of 'value' in optimizer.set_lr must be float, but received %s.zhoptimizer's learning rate can't be LRScheduler when invoke this API, because this will lead to conflict.NÚfill_constantÚOut)rj   r®   r§   T)rX   ÚoutputsÚattrsr»   )rR   Úintrb   rV   rX   rh   r   r¬   rÂ   r   r	   r   Úfull_rZ   r®   rj   r   r
   r$   Ú	append_op)r   r§   Ú
current_lrrà   r$   s        rD   Úset_lrzOptimizer.set_lr3  s  € ôF ˜%¤#¤u Ô.ÜØYÜ˜“;ñ óð ô �d×)Ñ)¬;Ô7ÜØzóð ô $ E›lˆÔØ×/Ñ/Ó1ˆ
ØÐ!ÜÔ Ü/Ó1�Ü—‘ØÜ˜×)Ñ)Ó*Ü˜%“LØ×$Ñ$Øõô  )×=Ñ=Ó?×LÑLÓN�Ø×&Ñ&Ø(Ø" Z LÐ1à!+×!1Ñ!1Ü!% j×&6Ñ&6Ó!7Ü!& u£ñð
 #'ð 'õ 	ð "rF   c                 ód   — ddl m} t        ||«      st        dt	        |«      z  «      ‚|| _        y)a  
        :api_attr: imperative

        Set the LRScheduler of the learning rate manually in the optimizer. If the optimizer already used LRScheduler previously,
        this API will set it be the new one.

        Args:
            scheduler (LRScheduler): the LRScheduler of learning rate

        Returns:
            None

        Examples:
            .. code-block:: python

                >>> import paddle
                >>> linear = paddle.nn.Linear(10, 10)

                >>> adam = paddle.optimizer.Adam(0.1, parameters=linear.parameters())

                >>> # set learning rate manually by class LRScheduler
                >>> scheduler = paddle.optimizer.lr.MultiStepDecay(learning_rate=0.5, milestones=[2,4,6], gamma=0.8)
                >>> adam.set_lr_scheduler(scheduler)
                >>> lr = adam.get_lr()
                >>> print("current lr is {}".format(lr))
                current lr is 0.5

                >>> # set learning rate manually by another LRScheduler
                >>> scheduler = paddle.optimizer.lr.StepDecay(learning_rate=0.1, step_size=5, gamma=0.6)
                >>> adam.set_lr_scheduler(scheduler)
                >>> lr = adam.get_lr()
                >>> print("current lr is {}".format(lr))
                current lr is 0.1

        r   r   zZThe type of 'scheduler' in optimizer.set_lr_schduler must be LRScheduler, but received %s.N)Úpaddle.optimizer.lrr   rR   rV   rX   rh   )r   Ú	schedulerr   s      rD   Úset_lr_schedulerzOptimizer.set_lr_schedulerx  s8   € õJ 	4ä˜) [Ô1ÜØlÜ˜	“?ñ$óð ð (ˆÕrF   c                 ón   — t        | j                  t        «      r| j                  S | j                  «       S )aô  
        Get current learning rate of optimizer.
        If 'LRScheduler' is not used, the return value is all the same.
        If 'LRScheduler' is used, the return value is the current scheduled learing rete.

        Returns:
            float: The current learning rate of optimizer.

        Examples:
            .. code-block:: python

                >>> # train on default dynamic graph mode
                >>> import paddle
                >>> import numpy as np
                >>> emb = paddle.nn.Embedding(10, 3)

                >>> ## example1: LRScheduler is not used, return the same value is all the same
                >>> adam = paddle.optimizer.Adam(0.01, parameters = emb.parameters())
                >>> for batch in range(10):
                ...     input = paddle.randint(low=0, high=5, shape=[5])
                ...     out = emb(input)
                ...     out.backward()
                ...     print("Learning rate of step{}: {}".format(batch, adam.get_lr())) # 0.01
                ...     adam.step()
                Learning rate of step0: 0.01
                Learning rate of step1: 0.01
                Learning rate of step2: 0.01
                Learning rate of step3: 0.01
                Learning rate of step4: 0.01
                Learning rate of step5: 0.01
                Learning rate of step6: 0.01
                Learning rate of step7: 0.01
                Learning rate of step8: 0.01
                Learning rate of step9: 0.01

                >>> ## example2: StepDecay is used, return the scheduled learning rate
                >>> scheduler = paddle.optimizer.lr.StepDecay(learning_rate=0.5, step_size=2, gamma=0.1)
                >>> adam = paddle.optimizer.Adam(scheduler, parameters = emb.parameters())
                >>> for batch in range(10):
                ...     input = paddle.randint(low=0, high=5, shape=[5])
                ...     out = emb(input)
                ...     out.backward()
                ...     print("Learning rate of step{}: {}".format(batch, adam.get_lr())) # 0.5->0.05...
                ...     adam.step()
                ...     scheduler.step()
                Learning rate of step0: 0.5
                Learning rate of step1: 0.5
                Learning rate of step2: 0.05
                Learning rate of step3: 0.05
                Learning rate of step4: 0.005000000000000001
                Learning rate of step5: 0.005000000000000001
                Learning rate of step6: 0.0005000000000000001
                Learning rate of step7: 0.0005000000000000001
                Learning rate of step8: 5.000000000000001e-05
                Learning rate of step9: 5.000000000000001e-05

                >>> # train on static graph mode
                >>> paddle.enable_static()
                >>> main_prog = paddle.static.Program()
                >>> start_prog = paddle.static.Program()
                >>> with paddle.static.program_guard(main_prog, start_prog):
                ...     x = paddle.static.data(name='x', shape=[None, 10])
                ...     z = paddle.static.nn.fc(x, 100)
                ...     loss = paddle.mean(z)
                ...     scheduler = paddle.optimizer.lr.StepDecay(learning_rate=0.5, step_size=2, gamma=0.1)
                ...     adam = paddle.optimizer.Adam(learning_rate=scheduler)
                ...     adam.minimize(loss)

                >>> exe = paddle.static.Executor()
                >>> exe.run(start_prog)
                >>> for batch in range(10):
                ...     print("Learning rate of step{}: {}".format(batch, adam.get_lr())) # 0.5->0.05->0.005...
                ...     out = exe.run(main_prog, feed={'x': np.random.randn(3, 10).astype('float32')})
                ...     scheduler.step()
                Learning rate of step0: 0.5
                Learning rate of step1: 0.5
                Learning rate of step2: 0.05
                Learning rate of step3: 0.05
                Learning rate of step4: 0.005000000000000001
                Learning rate of step5: 0.005000000000000001
                Learning rate of step6: 0.0005000000000000001
                Learning rate of step7: 0.0005000000000000001
                Learning rate of step8: 5.000000000000001e-05
                Learning rate of step9: 5.000000000000001e-05
        )rR   rh   rb   rˆ   s    rD   Úget_lrzOptimizer.get_lr¦  s0   € ôl �d×)Ñ)¬5Ô1Ø×&Ñ&Ð&à×&Ñ&Ó(Ð(rF   c                 ó¸   — |€=t        «       rt        j                  «       }nt        j                  j                  «       }| j
                  j                  |d«      S )zC
        get global decayed learning rate
        :return:
        N)r   r   r
   rS   rÃ   rk   r”   )r   r7   s     rD   rÂ   zOptimizer._global_learning_rate  sH   € ð
 ˆ?ÜÔ Ü#×8Ñ8Ó:‘ä Ÿ-™-×<Ñ<Ó>�Ø×&Ñ&×*Ñ*¨7°DÓ9Ð9rF   c                 ó   — t        d«      ‚)zFappend optimize operator to block and return all the added optimize_opzcClass "Optimizer" connot be used directly as an optimizer, please use its subclasses such as "Adam")ÚNotImplementedError)r   r#   Úparam_and_grads      rD   Ú_append_optimize_opzOptimizer._append_optimize_op  s   € ä!Øuó
ð 	
rF   c                 óŒ  — |d   }t        |d«      rö|j                  d   }t        |t        t        j
                  j                  f«      r|S |dk(  r| j                  «       S t        «       sjt        j                  j                  «       j                  d¬«      5  t        j                  d«      5  | j                  «       |z  cd d d «       cd d d «       S t        j                  d«      5  | j                  «       |z  cd d d «       S | j                  «       S # 1 sw Y   nxY w	 d d d «       y # 1 sw Y   y xY w# 1 sw Y   y xY w)Nr   Úoptimize_attrr€   ç      ð?T)Úis_with_optÚscale_with_param_lr)r^   rû   rR   r   rS   rÈ   rË   rÂ   r   rÃ   r
   Ú_lr_schedule_guardr   r   )r   rø   rC   Úparam_lrs       rD   Ú_create_param_lrzOptimizer._create_param_lr  s!  € à˜qÑ!ˆÜ�5˜/Ô*Ø×*Ñ*¨?Ñ;ˆHÜ˜(¤X¬v¯z©z×/BÑ/BÐ$CÔDØ�à˜s’?Ø×5Ñ5Ó7Ð7ä&œ=Ü#Ÿ]™]×?Ñ?ÓA×TÑTØ(,ð Uõ ä$×/Ñ/Ø1õð $(×#=Ñ#=Ó#?À(Ñ#J÷ð ÷ñ ô '×1Ñ1Ð2GÕHØ#'×#=Ñ#=Ó#?À(Ñ#J÷ IÑHð ×-Ñ-Ó/Ð/÷ð úð ÷÷ ñ ú÷ IÐHús0   ÂD.Â0DÃ	D.Ã+D:ÄD!	ÄD.Ä.D7Ä:Ec                 óÖ  — |j                   | j                  v r| j                  |j                      }|S | j                  |«      }t        «       rt	        j
                  |d«      }nêt        j                  «       rt	        j
                  |d«      }||_         n¸t        | j                  t        «      sJ ‚t        j                  j                  ||j                  ddd¬«      }| j                  j                  j                  «       }|j!                  dd|gid|gi|j"                  t$        j&                  j(                  j*                  dœ¬	«       || j                  |j                   <   |S )
NrÁ   r   Tr¿   ÚcastÚXrç   )Úin_dtypeÚ	out_dtype)rX   Úinputsrè   ré   )r‚   r}   Ú_gen_master_weight_var_namer   rS   r  r   r   rR   rm   r   rÃ   rÖ   r®   rØ   r$   rì   rj   r   ÚVarDescÚVarTypeÚFP32)r   rC   r?   Úvar_namer#   s        rD   Ú_create_master_weightzOptimizer._create_master_weight.  s3  € Ø�:‰:˜×-Ñ-Ñ-Ø×&Ñ& u§z¡zÑ2ˆCð8 ˆ
ð5 ×7Ñ7¸Ó>ˆHÜŒ}Ü—k‘k %¨Ó3‘Ü×*Ñ*Ô,Ü—k‘k %¨Ó3�Ø#�•ä! $§+¡+¬{Ô;Ð;Ð;Ü—m‘m×5Ñ5Ø!ØŸ+™+ØØ#Ø $ð 6ó �ð Ÿ™×3Ñ3×@Ñ@ÓB�Ø—‘ØØ % ˜>Ø" S E˜Nà$)§K¡KÜ%)§\¡\×%9Ñ%9×%>Ñ%>ñð	  ô ð 03ˆD× Ñ  §¡Ñ,Øˆ
rF   c                 óJ   — |j                   dz   }t        j                  |«      S )NÚ_fp32_master)r‚   r   rÅ   )r   rC   r  s      rD   r  z%Optimizer._gen_master_weight_var_nameN  s!   € Ø—:‘: Ñ.ˆÜ×#Ñ# HÓ-Ð-rF   c           	      ó´  — | j                  |j                  «      sJ ‚|j                  | j                  v r| j                  |j                     }|S |j                  dz   }t	        j
                  |«      }|j                  j                  ||j                  dd|j                  |j                  |j                  ¬«      }|| j                  |j                  <   |S )Nr  r   rÁ   )r‚   r®   r§   rj   Ú	lod_levelrº   Úis_data)Ú_is_dtype_fp16_or_bf16rj   r‚   r†   r   rÅ   r#   Ú
create_varr®   r  rº   r  )r   Úgradr?   r  s       rD   Ú_create_master_gradzOptimizer._create_master_gradR  sÂ   € Ø×*Ñ*¨4¯:©:Ô6Ð6Ð6Ø�9‰9˜×*Ñ*Ñ*Ø×$Ñ$ T§Y¡YÑ/ˆCð ˆ
ð —y‘y >Ñ1ˆHÜ"×+Ñ+¨HÓ5ˆHØ—*‘*×'Ñ'ØØ—j‘jØØØŸ.™.Ø ×,Ñ,ØŸ™ð (ó ˆCð -0ˆD×Ñ˜tŸy™yÑ)Øˆ
rF   c                  ó   — y)zÍCreate all accumulators needed by the parameters

        Args:
            block: the block in which the loss tensor is present
            parameters: list of parameter tensors for the optimizer
        NrN   )r   r#   r�   s      rD   Ú_create_accumulatorszOptimizer._create_accumulatorse  ó   € ð 	rF   c                  ó   — y)a  Finish any custom updates needed
           before completing an optimization step

        Args:
            block: the block in which the loss tensor is present
            parameters: list of parameter tensors for the optimizer

        Returns:
            None
        NrN   )r   r#   Úparameters_and_gradss      rD   Ú_finish_updatezOptimizer._finish_updaten  s   € ð 	rF   c           
      ó~  — | j                   �| j                   dz   |z   }|| j                  v rf|j                  | j                  |   v rKt        j                  «       r| j                  |   |j                     S t        d|› d|j                  › �«      ‚|€|j                  }|j                  dz   |z   }t        j                  |«      }| j                  j                  |«       |€| j                  |j                  «      }t        «       rnt        j                  j                  j!                  |xs |j"                  ||t        j$                  j&                  j)                  t+        |«      ¬«      ¬«      }	�nt-        | j.                  t0        «      sJ ‚| j.                  j3                  |d|xs |j"                  t        j4                  j6                  j8                  |d¬«      }	t	        «       r…|d	k(  st-        |t        j:                  «      rft        j<                  «       sRt?        j@                  |	|	j                  tC        t+        |«      «      |	j"                  t        j:                  «       «       nbtE        |«      5  | j.                  jG                  |	t        j$                  j&                  j)                  t+        |«      ¬«      ¬«       ddd«       t        j                  «       r«tI        | jJ                  «      d
kD  r“|| jJ                  v sJ d|› d�«       ‚|	jM                  | jJ                  jO                  |«      «       t        j<                  «       r<|	jQ                  «       jS                  | jJ                  jU                  |dz   d«      «       |	| j                  |   |j                  <   |	S # 1 sw Y   ŒæxY w)a|  Utility function to add an accumulator for a parameter

        Args:
            block: the block in which the loss tensor is present
            name: name of the accumulator
            param: parameter tensor for which accumulator is to be added
            dtype: data type of the accumulator tensor
            fill_value: value to initialize the accumulator tensor
        Nr’   úAccumulator z already exists for parameter r¹   r¼   T)r‚   rº   rj   rX   r®   Úbelong_to_optimizerÚcpur   zOptimizer set error, z should in state dictr—   r¤   )+r\   rl   r‚   r   r   Ú	Exceptionr®   r   rÅ   rn   r+   Ú_get_device_for_paramr   rS   rÈ   r   rÔ   rj   rc   r½   rÇ   rb   rR   rm   r   rÏ   r	  r
  Ú
LOD_TENSORÚCPUPlacer›   r   rë   r­   r   rÐ   r/   ro   Ú	set_valuer¦   rœ   r¨   r”   )
r   r‚   rC   rj   Ú
fill_valuer®   rX   Údevicer  r?   s
             rD   Ú_add_accumulatorzOptimizer._add_accumulator{  s  € ð& �:‰:Ð!Ø—:‘: Ñ# dÑ*ˆDà�D×&Ñ&Ñ&Ø—
‘
˜d×0Ñ0°Ñ6Ñ6ä×(Ñ(Ô*Ø×)Ñ)¨$Ñ/°·
±
Ñ;Ð;ÜØ˜t˜fÐ$BÀ5Ç:Á:À,ÐOóð ð ˆ=Ø—K‘KˆEà—:‘: Ñ# dÑ*ˆÜ×'Ñ'¨Ó1ˆØ×Ñ×#Ñ# HÔ-àˆ>Ø×/Ñ/°·
±
Ó;ˆFäŒ=Ü—*‘*—/‘/×2Ñ2ØÒ$˜Ÿ™ØØÜ"ŸI™I×1Ñ1×:Ñ:Ü 
Ó+ð ;ó ð	 3ó ŠCô ˜dŸk™k¬;Ô7Ð7Ð7Ø—+‘+×4Ñ4ØØ ØÒ*˜uŸ{™{Ü—\‘\×)Ñ)×4Ñ4ØØ$(ð 5ó ˆCô  Ô!Ø˜u’_¬
°6¼4¿=¹=Ô(IÜ×2Ñ2Ô4ä—‘ØØ—I‘IÜœ˜jÓ)Ó*Ø—I‘IÜ—M‘M“Oõô " &Õ)Ø—K‘K×8Ñ8ØÜ$*§I¡I×$9Ñ$9×$BÑ$BÜ"'¨
Ó"3ð %Có %ð 9ô ÷ *ô ×(Ñ(Ô*Ü�t×0Ñ0Ó1°AÒ5à  D×$=Ñ$=Ñ=ðOà.¨x¨jÐ8MÐNóOØ=à—M‘M $×";Ñ";×"?Ñ"?ÀÓ"IÔJô ×0Ñ0Ô2ØŸ™Ó(×<Ñ<Ø ×5Ñ5×9Ñ9Ø (¨>Ñ 9¸4óôð 03ˆ×Ñ˜4Ñ  §¡Ñ,Øˆ
÷1 *Ð)ús   É?AN3Î3N<c                 óþ   — | j                   �| j                   dz   |z   }|| j                  vs|j                  | j                  |   vrt        d|› d|j                  › �«      ‚| j                  |   |j                     S )a  Utility function to fetch an accumulator for a parameter

        Args:
            name: name of the accumulator
            param: parameter tensor for which accumulator is to be fetched

        Returns:
            accumulator tensor for the parameter
        r’   r  ú does not exist for parameter )r\   rl   r‚   r!  )r   r‚   rC   s      rD   Ú_get_accumulatorzOptimizer._get_accumulatorÞ  s„   € ð �:‰:Ð!Ø—:‘: Ñ# dÑ*ˆDà˜×*Ñ*Ñ*Ø�z‰z ×!3Ñ!3°DÑ!9Ñ9äØ˜t˜fÐ$BÀ5Ç:Á:À,ÐOóð ð ×!Ñ! $Ñ'¨¯
©
Ñ3Ð3rF   c                 óf  — | j                   �| j                   dz   |z   }| j                  xr | j                  |j                  «      }|r| j                  |j
                     n|}|j
                  }|| j                  vs|| j                  |   vrt        d|› d|› �«      ‚| j                  |   |   S )a
  Utility function to fetch an accumulator for a parameter
        Args:
            name: name of the accumulator
            param: parameter variable for which accumulator is to be fetched
        Returns:
            accumulator variable for the parameter
        r’   r  r*  )r\   Ú_multi_precisionr  rj   r}   r‚   rl   r!  )r   r‚   rC   Úfind_masterÚtarget_paramÚtarget_names         rD   Ú_get_accumulator_masterz!Optimizer._get_accumulator_masteró  sÆ   € ð �:‰:Ð!Ø—:‘: Ñ# dÑ*ˆDØ×+Ñ+ò 
°×0KÑ0KØ�K‰Kó1
ˆñ 1<ˆD× Ñ  §¡Ò,Àð 	ð #×'Ñ'ˆà˜×*Ñ*Ñ*Ø $×"4Ñ"4°TÑ":Ñ:äØ˜t˜fÐ$BÀ;À-ÐPóð ð ×!Ñ! $Ñ'¨Ñ4Ð4rF   c                 ó  — |D ]„  }|d   j                   du sŒ|d   j                  }|j                  }t        j                  j                  «       }|D ]2  }|j                  }||v sŒ|j                  |«      | j                  |<    Œ„ Œ† y )Nr   F)	r»   r‚   r(   r   Úop_proto_and_checker_makerÚkOpDeviceAttrNameÚinput_arg_namesÚattrrp   )	r   r  Útarget_blockrø   Ú
param_namer(   Údevice_attr_namer*   r5  s	            rD   Ú_update_param_device_mapz"Optimizer._update_param_device_map  s‘   € Û2ˆNØ˜aÑ ×.Ñ.°%Ò7Ø+¨AÑ.×3Ñ3�
Ø"×&Ñ&�ä×3Ñ3×EÑEÓGð !ó �BØ&(×&8Ñ&8�OØ! _Ò4Ø=?¿W¹WØ,ó>˜×.Ñ.¨zÑ:ñ ñ ñ 3rF   c                 óD   — d }|| j                   v r| j                   |   }|S rM   )rp   )r   r8  r'  s      rD   r"  zOptimizer._get_device_for_param  s*   € ØˆØ˜×/Ñ/Ñ/Ø×+Ñ+¨JÑ7ˆFØˆrF   c           	      óŽ  — t        j                  «       j                  «       }|}t        j                  «       j                  «       }|j                  |j                  k7  rA|j
                  dk7  sJ d«       ‚t        j                  «       j                  |j
                     }t        |j                  «      }t        | j                  j                  «      | _        | j                  «        | j                  �r÷| j                  j                  dv �rÞt        | j                  d   |   «      dk(  r°t        | j                  d   |   «      dk(  r’t!        |t"        «      r;|dk(  sJ ‚| j%                  ||D �cg c]  }|d   j&                  s|d   ‘Œ c}|«       nG| j)                  |«       | j%                  ||d   D �cg c]  }|d   j&                  s|d   ‘Œ c}|«       t        j*                  «       r| j-                  |||¬«       �n| j/                  ||«       g }|D ]@  }	|	d   j&                  rŒ|	d	   €Œ|j1                  |	d   «       |j1                  |	d	   «       ŒB |d   j2                  j4                  j7                  |«      5  t9        d«      5  | j;                  |d   j<                  «      }
t?        |
«      5  | j-                  |||¬«       d
d
d
«       d
d
d
«       d
d
d
«       �n!t        j*                  «       s)t!        |t@        «      r|d   n|}| j/                  ||«       t!        |t"        «      rdtB        jD                  j                   jG                  «       5  | jI                  ||D �cg c]  }|d   j&                  s|d   ‘Œ c}«       d
d
d
«       n{|jK                  «       }|d   D �cg c]  }|d   j&                  s|d   ‘Œ c}|d<   tB        jD                  j                   jG                  «       5  | jI                  ||«       d
d
d
«       t        j*                  «       �r9| jM                  d«      }|r9t!        |tN        jP                  jR                  «      �r¨| jU                  dd«       �n”t!        |tN        jP                  jR                  «      r| jU                  dd«       t!        |t"        «      r3|D ],  }	|	d	   €Œ	|	d   j&                  du sŒ| jW                  ||	«       Œ. �n|d   D ]k  }	|	d	   €Œ	|	d   j&                  du sŒi }|	|d<   |jY                  |j[                  «       D ��ci c]  \  }}|dk7  r||“Œ c}}«       | jW                  ||«       Œm n§|D ]¢  }	|	d	   €Œ	|	d   j2                  j4                  j7                  |	«      5  t9        d«      5  |	d   j&                  du rD| j;                  |	d   j<                  «      }
t?        |
«      5  | jW                  ||	«      }d
d
d
«       d
d
d
«       d
d
d
«       Œ¤ | j]                  ||«       t        |j                  «      }|j_                  ||«      S c c}w c c}w # 1 sw Y   �Œ€xY w# 1 sw Y   �Œ…xY w# 1 sw Y   ŒfxY wc c}w # 1 sw Y   �ŒmxY wc c}w # 1 sw Y   �ŒxY wc c}}w # 1 sw Y   Œ®xY w# 1 sw Y   Œ²xY w# 1 sw Y   �ŒZxY w)áè  Add optimization operators to update gradients to tensors.

        Args:
          parameters_and_grads(list(tuple(Tensor, Tensor))):
            a list of (tensor, gradient) pair to update.

        Returns:
          return_op_list: a list of operators that will complete one step of
            optimization. This will include parameter update ops, global step
            update ops and any other custom ops required by subclasses to manage
            their internal state.
        éÿÿÿÿzFcurrent block is not global_block, but it doesn't have backward block.)ÚMomentumÚAdamrŽ   r   r�   rK   ©Úparam_group_idxr   NÚ	optimizerÚ	found_infTF)0r   r
   r$   r"   ÚidxÚbackward_block_idxÚblocksr/   r(   r   Ú	__class__Ú__name__rm   rä   rw   ry   rR   rZ   Ú_multi_tensor_initr»   Ú_update_param_groupr   Ú _append_optimize_multi_tensor_opr:  r+   r#   r7   Ú_optimized_guardr   r"  r‚   r   rY   rS   rÑ   rã   r  rv   r•   r   rU   rT   rŒ   rù   Úupdaterš   r  Ú
_slice_ops)r   r  rB  r$   r7  r"   ÚstartÚpÚparam_grad_listrø   r'  Úparams_grads_device_mapÚparams_acc_dictrD  Úparam_grad_dictrŸ   r    Úoptimize_opÚends                      rD   Ú_create_optimization_passz#Optimizer._create_optimization_pass#  s¢  € ô4 !×5Ñ5Ó7×DÑDÓFˆØ#ˆÜ!×6Ñ6Ó8×FÑFÓHˆØ×Ñ × 0Ñ 0Ò0à×0Ñ0°BÒ6ðXàWóXØ6ä$×9Ñ9Ó;×BÑBØ×0Ñ0ñˆLô �L×$Ñ$Ó%ˆÜ! $§.¡.×"9Ñ"9Ó:ˆŒà×)Ñ)Ô+ð ×!Ó! d§n¡n×&=Ñ&=ð B
ò '
ô
 �D×$Ñ$Ð%5Ñ6°ÑGÓHÈAÒMÜ˜×(Ñ(Ð)9Ñ:¸?ÑKÓLØòô Ð2´DÔ9Ø*¨aÒ/Ð/Ð/Ø×+Ñ+Ø$ñ &:óá%9 Ø#$ Q¡4×#5Ò#5ð ˜a›DØ%9ñð
 (õð ×,Ñ,Ð-AÔBØ×+Ñ+Ø$ð &:¸(Ò%Cóá%C Ø#$ Q¡4×#5Ò#5ð ˜a›DØ%Cñð
 (ôô ×(Ñ(Ô*Ø×5Ñ5Ø Ø(Ø$3ð 6ö ð ×-Ñ-Ø(¨,ôð
 #%�Û&:�Nà*¨1Ñ-×;Ó;Ø*¨1Ñ-Ñ9à'×.Ñ.¨~¸aÑ/@ÔAØ'×.Ñ.¨~¸aÑ/@ÕAð ';ð % QÑ'×-Ñ-×5Ñ5×FÑFØ#õä˜kÕ*Ø!×7Ñ7¸ÈÑ8J×8OÑ8OÓP�FÜ% fÕ-Ø×=Ñ=Ø(Ø0Ø,;ð >ô ÷ .÷ +÷ñ ô ×,Ñ,Ô.ô "Ð"6¼Ô=ð )¨Ò2à-ð (ð
 ×-Ñ-Ø+¨\ôô Ð.´Ô5Ü—[‘[×*Ñ*×GÑGÕIØ×-Ñ-Ø$ñ &:óá%9 Ø#$ Q¡4×#5Ò#5ð ˜a›DØ%9ñô÷ JÐIð #7×";Ñ";Ó"=�ð -¨XÒ6ó-á6˜Ø˜Q™4×-Ò-ð �a“DØ6ñ-� Ñ)ô
 —[‘[×*Ñ*×GÑGÕIØ×-Ñ-¨l¸OÔL÷ Jô ×(Ñ(Õ*Ø ×3Ñ3°KÓ@�	ÙÜ! )¬T¯Z©Z×->Ñ->Õ?Ø×/Ñ/°¸TÖBä! )¬T¯Z©Z×->Ñ->Ô?Ø×/Ñ/°¸UÔCÜ!Ð"6¼Ô=Û.B˜NØ-¨aÑ0Ð8Ø (Ø-¨aÑ0×>Ñ>À%ÒGØ $× 8Ñ 8Ø$0°.õ!"ò	 /Cð /CÀ8Ô.L˜NØ-¨aÑ0Ð8Ø (Ø-¨aÑ0×>Ñ>À%ÒGØ24 Ø<J °Ñ 9Ø /× 6Ñ 6ð 5I×4NÑ4NÔ4Pô%&á4P©D¨A¨qØ+,°ª=ð )*¨1©Ø4Pò%&ô!"ð !%× 8Ñ 8Ø$0°/õ!"ñ /Mó" ';�NØ% aÑ(Ð0Ø Ø'¨Ñ*×0Ñ0×8Ñ8×IÑIØ&õä! +Õ.Ø)¨!Ñ,×:Ñ:¸eÑCØ%)×%?Ñ%?Ø .¨qÑ 1× 6Ñ 6ó&˜Fô ".¨fÕ!5Ø.2×.FÑ.FØ$0°.ó/" ÷ "6÷ /÷ð ð ';ð" 	×Ñ˜LÐ*>Ô?ä�,×"Ñ"Ó#ˆØ×&Ñ& u¨cÓ2Ð2ùòKùò÷> .Ñ-ú÷ +Ñ*ú÷ð üò2÷ JÑIüò-÷
 JÑIüó6%&÷( "6Ð!5ú÷ /Ð.ú÷ñ úsº   Å3Y
Æ;Y
Ê	Y,Ê*YÊ?YËYËY,Í#Y=Í4Y8ÎY=Î6Z
Ï?ZÔ9ZÖZ:Ö';Z.×"Z"	×5Z.×=Z:ÙYÙYÙY)	Ù$Y,Ù,Y5Ù8Y=Ù=ZÚZÚ"Z+Ú'Z.Ú.Z7Ú3Z:Ú:[	c           	      óþ  — t        j                  «       j                  «       }|}t        |j                  «      }| j                  «        t        |t        «      r|d   n|}| j                  ||«       t        |t        «      r4| j                  ||D �cg c]  }|d   j                  rŒ|d   ‘Œ c}«       nJ|j                  «       }|d   D �cg c]  }|d   j                  s|d   ‘Œ c}|d<   | j                  ||«       t        |t        «      r2|D ],  }	|	d   €Œ	|	d   j                  du sŒ| j                  ||	«       Œ. ns|d   D ]k  }	|	d   €Œ	|	d   j                  du sŒi }
|	|
d<   |
j                  |j                  «       D ��ci c]  \  }}|dk7  r||“Œ c}}«       | j                  ||
«       Œm | j!                  ||«       t        |j                  «      }|j#                  ||«      S c c}w c c}w c c}}w )r=  rK   r   r   F)r   r
   r$   r/   r(   rä   rR   rY   r:  rZ   r  r»   rv   rù   rN  rš   r  rO  )r   r  rB  r$   r7  rP  rS  rQ  rT  rø   rU  rŸ   r    rW  s                 rD   Ú_pir_create_optimization_passz'Optimizer._pir_create_optimization_passâ  s.  € ô  !×5Ñ5Ó7×DÑDÓFˆØ#ˆä�L×$Ñ$Ó%ˆà×)Ñ)Ô+ô Ð.´Ô5ð ! Ò*à%ð 	 ð
 	×%Ñ%Ð&=¸|ÔLäÐ*¬DÔ1Ø×%Ñ%ØÙ3ÓNÑ3˜!¸1¸Q¹4×;MÓ;M��1“Ð3ÑNõð
 3×7Ñ7Ó9ˆOð )¨Ò2ó)á2�AØ˜‘t×)Ò)ð �!“Ø2ñ)ˆO˜HÑ%ð
 ×%Ñ% l°OÔDäÐ*¬DÔ1Û"6�Ø! !Ñ$Ð,ØØ! !Ñ$×2Ñ2°eÒ;Ø×,Ñ,¨\¸>ÕJñ	 #7ð #7°xÔ"@�Ø! !Ñ$Ð,ØØ! !Ñ$×2Ñ2°eÒ;Ø&(�OØ0>�O HÑ-Ø#×*Ñ*ð )=×(BÑ(BÔ(Dôá(D¡  1Ø  Hš}ð ˜q™DØ(Dòôð ×,Ñ,¨\¸?ÕKð #Að" 	×Ñ˜LÐ*>Ô?ä�,×"Ñ"Ó#ˆØ×&Ñ& u¨cÓ2Ð2ùòM Oùò)ùó*s   ÂG/
Â(G/
ÃG4ÆG9c                 óÌ  — d}t        j                  «       rn| j                  ||«      }| j                  €|j                  | _        t        j                  «       r_|r|n| j
                  }g }t        j                  j                  |«      }	t        |	«      D ]  \  }
}|€Œ	|j                  ||
   |f«       Œ  |S |€&t        j                  j                  j                  g}nt        |t         «      sJ ‚|j"                  j$                  }t'        j(                  |j*                  «      dk(  s J dj-                  |j*                  «      «       ‚|r|n| j
                  }t        j.                  j1                  ||«      5  t3        «       r˜|€;|j5                  «       j7                  «       }|D �cg c]  }|j8                  du r|‘Œ }}g }t        j:                  j<                  j?                  |||¬«      }	t        |	«      D ]  \  }
}|€Œ	|j                  ||
   |f«       Œ  n+ddl m!}  |«       rtE        |g|||«      }ntG        ||||«      }ddd«       |S c c}w # 1 sw Y   S xY w)aÝ  
        The first part of ``minimize``, do auto-diff to append backward operations for
        the current program.

        Args:
            loss (Tensor): ``loss`` tensor to run optimizations.
            startup_program (Program, optional): :ref:`api_paddle_static_Program` for
                initializing parameters in ``parameters``. The default value
                is None, at this time :ref:`api_paddle_static_default_startup_program` will be used.
            parameters (list, optional): List of ``Tensor`` or ``Tensor.name`` to update
                to minimize ``loss``. The default value is None, at this time all parameters
                will be updated.
            no_grad_set (set, optional): Set of ``Tensor``  or ``Tensor.name`` that don't need
                to be updated. The default value is None.
            callbacks (list, optional): list of callable objects to run when appending backward
                operator for one parameter. The default value is None.

        Return:
            list: list of (param, grad) tensor pairs, param is ``Parameter``,
                grad is the gradient value corresponding to the parameter.

        Examples:
            .. code-block:: python

                >>> import paddle
                >>> x = paddle.arange(26, dtype="float32").reshape([2, 13])

                >>> linear = paddle.nn.Linear(13, 5)
                >>> # This can be any optimizer supported by dygraph.
                >>> adam = paddle.optimizer.Adam(learning_rate = 0.01,
                ...                             parameters = linear.parameters())
                >>> out = linear(x)
                >>> out.backward()
                >>> adam.step()
                >>> adam.clear_grad()
        Nr   z´The number of elements of loss should be 1, but the current loss.shape is {}, whose number of elements is not 1. Maybe that you should call paddle.mean to process the current loss.F)Úno_grad_varsr   )Úprim_enabled)$r   r   Ú_get_no_grad_setri   rj   r[   r   rU   Úget_all_gradsr0   r+   rS   rc   rd   Úerror_clip_callbackrR   rZ   r#   r7   r©   Úprodr®   rW   rÃ   rÆ   r   r$   r%   r»   ÚautogradÚir_backwardr  Úpaddle.incubate.autograd.utilsr]  rE   r   )r   ÚlossrØ   r�   r3   r4   Úact_no_grad_setr2   Úparams_gradsÚgradsr)   r  r7   Úprogram_all_paramsrC   r]  s                   rD   ÚbackwardzOptimizer.backward+  s^  € ðX ˆÜ×$Ñ$Ô&Øà"×3Ñ3°D¸+ÓFˆOð �;‰;ÐØŸ*™*ˆDŒKä×$Ñ$Ô&Ù+5™Z¸4×;OÑ;OˆNð ˆLÜ—J‘J×,Ñ,¨^Ó<ˆEÜ(¨Ö/‘��tØÑ#Ø ×'Ñ'¨¸Ñ)>ÀÐ(EÕFð  0ð\ ÐðU Ð Ü#ŸY™YŸ^™^×?Ñ?Ð@‘	ä! )¬TÔ2Ð2Ð2Ø—j‘j×(Ñ(ˆGÜ—7‘7˜4Ÿ:™:Ó&¨!Ò+ð ðVßV\ÑV\Ø—J‘JóWóÐ+ñ ,6™Z¸4×;OÑ;OˆNÜ—‘×,Ñ,¨W°oÕFÜ”=Ø%Ð-ð $×0Ñ0Ó2×AÑAÓCð +ñ
 *<ó*á); Ø$×2Ñ2°eÑ;ò "Ø);ð 'ð *ð
 $&�LÜ"ŸO™O×7Ñ7×<Ñ<Ø˜n¸?ð =ó �Eô (1°Ö'7™˜˜tØÑ+Ø(×/Ñ/°ÀÑ1FÈÐ0MÕNñ (8õ Lá#”~Ü':Ø!˜F N°OÀYó(™ô (7Ø  .°/À9ó(˜÷7 Gð< Ðùò/*÷ Gð< Ðús&   Å</IÆ+IÇAIÈAIÉIÉI#c                 ó"  — t        | d«      st        |d„ ¬«      }| j                  �| j                  |«      }n)t        j                  j
                  j                  |«      }| j                  || j                  «      }| j                  |«      }|S )ad  
        Second part of `minimize`, appending optimization operators for
        given `params_grads` pairs.

        Args:
            params_grads (list): list of (param, grad) pair to do optimization.

        Returns:
            list: A list of operators appended to the current program.

        Examples:
            .. code-block:: python

                >>> import paddle

                >>> inp = paddle.uniform([10, 10], dtype="float32", min=-0.1, max=0.1)
                >>> linear = paddle.nn.Linear(10, 10)
                >>> out = linear(inp)
                >>> loss = paddle.mean(out)
                >>> optimizer = paddle.optimizer.Adam(learning_rate=0.1,
                ...         parameters=linear.parameters())
                >>> params_grads = optimizer.backward(loss)
                >>> optimizer.apply_gradients(params_grads)

        Ú_sortedc                 ó    — | d   j                   S ©Nr   )r‚   )Úxs    rD   rO   z+Optimizer.apply_gradients.<locals>.<lambda>´  s   € ¸aÀ¹d¿iºirF   )rŠ   )
r^   r-   rg   rS   rc   rd   Úappend_gradient_clip_opsÚappend_regularization_opsrf   rX  )r   rg  Úoptimize_opss      rD   Úapply_gradientszOptimizer.apply_gradients˜  s‚   € ô6 �t˜YÔ'Ü! ,Ñ4GÔHˆLð �?‰?Ð&ØŸ?™?¨<Ó8‰Lä!Ÿ9™9Ÿ>™>×BÑBÀ<ÓPˆLð ×5Ñ5Ø˜$×-Ñ-ó
ˆð ×5Ñ5°lÓCˆØÐrF   c                 ó^  — t        j                  «       rt        ryt        «       �rt        j
                  j                  t        j
                  j                  «       t        j
                  j                  «       «      5  t        |t        «      r:| j                  �| j                  |«      }| j                  || j                  «      }n7|d   }|� ||d   «      |d<   | j                  |d   | j                  «      |d<   t        «       r| j                  ||¬«      }n| j!                  ||¬«      }ddd«       |S |dk(  sJ ‚|j"                  j$                  }t        j
                  j                  ||«      5  | j'                  |«      }ddd«       |S # 1 sw Y   S xY w# 1 sw Y   S xY w)aÜ  
        Second part of `minimize`, appending optimization operators for
        given `params_grads` pairs.
        Args:
            loss (Tensor): loss tensor to run optimizations.
            startup_program (Program): startup_program for initializing parameters
                in `parameters`.
            params_grads (list): list of (param, grad) pair to do optimization.
        Returns:
            list: A list of operators appended to the current program.
        NrQ   rK   rA  r   )r   r   Ú g_shard_bypass_dygraph_optimizerr   rS   rÃ   rÆ   r
   rÄ   rR   rZ   rg   rq  rf   r   rZ  rX  r#   r7   rs  )r   re  rØ   rg  rB  rQ   rr  r7   s           rD   Ú_apply_optimizezOptimizer._apply_optimizeÄ  s–  € ô ×$Ñ$Ô&Õ+KØä!Õ#Ü—‘×,Ñ,Ü—‘×2Ñ2Ó4Ü—‘×5Ñ5Ó7õô ˜l¬DÔ1Ø—‘Ð2Ø'+§¡°|Ó'D˜Ø#'×#AÑ#AØ$ d×&9Ñ&9ó$‘Lð !-¨[Ñ 9�IØ Ð,Ù1:Ø(¨Ñ2ó2˜ XÑ.ð .2×-KÑ-KØ$ XÑ.°×0CÑ0Có.�L Ñ*ô ”=Ø#'×#EÑ#EØ$°oð $Fó $‘Lð $(×#AÑ#AØ$°oð $Bó $�L÷3ðB Ðð	 # aÒ'Ð'Ð'Ø—j‘j×(Ñ(ˆGÜ—‘×,Ñ,¨W°oÕFØ#×3Ñ3°LÓA�÷ GàÐ÷CðB Ðú÷ GàÐús   Á?B3FÅ9F"ÆFÆ"F,c                 ó  ‡ — |�&t        |d«      rt        |d«      r|j                  €|€|S d}ˆ fd„} |||«      }t        |d«      r*|j                  �|j                  |||j                  «      }n|� ||||j                  «      }|€J ‚t        «       rt	        j
                  ||g«      S |}|j                  t        j                  j                  j                  k(  r|j                  j                  |j                  t        j                  «       z   |j                  |j                  |j                   t        j                  j                  j"                  ¬«      }d||gi}d|gi}|j                  j%                  d||¬«       |S )	zpCreate and add backward regularization Operators

        Function helper of append_regularization_ops.
        NrJ   c                 ó0  •— | }| j                   |j                   k7  ry‰j                  xr ‰j                  | j                   «      }|r3t        ‰j                  «      dk7  r‰j                  | j
                     }|S | j                  |j                   «      }|S rn  )rj   r-  r  r/   r}   r‚   Úastype)rC   r  r/  r.  r   s       €rD   Úget_target_paramzBOptimizer._create_regularization_of_grad.<locals>.get_target_param  s‹   ø€ Ø ˆLØ�{‰{˜dŸj™jÒ(à×)Ñ)ò AØ×3Ñ3°E·K±KÓ@ð ñ ¤3 t×';Ñ';Ó#<ÀÒ#AØ#'×#7Ñ#7¸¿
¹
Ñ#C�Lð  Ðð $)§<¡<°·
±
Ó#;�LØÐrF   )r‚   rj   r®   r  rX   r  rç   Úsum)rX   r  rè   )r^   rJ   r#   r   r   Úadd_nrX   r   r	  r
  ÚSELECTED_ROWSr  r‚   ÚkNewGradSuffixrj   r®   r  r#  rì   )	r   rC   r  rf   Úregularization_termrz  Únew_gradr  rè   s	   `        rD   Ú_create_regularization_of_gradz(Optimizer._create_regularization_of_gradú  sp  ø€ ð ˆ<ä˜E =Ô1Ü˜E =Ô1°e×6GÑ6GÐ6OàÐ&àˆKØ"Ðô	 ñ ! ¨Ó-ˆÜ�5˜-Ô(¨U×->Ñ->Ð-Jà"'×"3Ñ"3°E¸4ÀÇÁÓ"LÑØÐ'Ù"0°¸¸d¿j¹jÓ"IÐà"Ð.Ð.Ð.ä!Ô#Ü—<‘< Ð':Ð ;Ó<Ð<àˆHØ�y‰yœDŸL™L×0Ñ0×>Ñ>Ò>ð
  Ÿ:™:×0Ñ0ØŸ™¤T×%8Ñ%8Ó%:Ñ:ØŸ+™+ØŸ+™+Ø#Ÿo™oÜŸ™×-Ñ-×8Ñ8ð 1ó �ð ˜DÐ"5Ð6Ð7ˆFØ˜x˜jÐ)ˆGØ�J‰J× Ñ  e°FÀGÐ ÔLàˆOrF   c                 óN  — g }t        j                  «       s
t        «       r2|D ]+  \  }}| j                  |||«      }|j	                  ||f«       Œ- |S d}t        j
                  d«      5  |D ]“  \  }}|s6|j                  �*|�(d}t        j                  d|j                  «       z  «       |j                  j                  j                  ||g«      5  | j                  |||«      }|j	                  ||f«       ddd«       Œ• 	 ddd«       |S # 1 sw Y   ŒªxY w# 1 sw Y   |S xY w)ab  Create and add backward regularization Operators

        Creates and adds backward regularization operators in the BlockDesc.
        This will add gradients of the regularizer function to the gradients
        of the parameters and return these modified gradients. This is the
        same as implementing weight decay in optimizers for regularization.

        Args:
            parameters_and_grads: A list of (parameters, gradients) pairs
                                  that need to be regularized.
            regularization: A global regularizer. If the parameter is not
                            set. It will be applied with regularizer.

        Returns:
            list[(Variable, Variable)]: list of (parameters, gradients) \
            pair with the regularized gradient

        Raises:
            Exception: Unknown regularization type
        Frf   NTzÐIf regularizer of a Parameter has been set by 'base.ParamAttr' or 'base.WeightNormParamAttr' already. The Regularization[%s] in Optimizer will not take effect, and it will only be applied to other Parameters!)r   r   r   r�  r+   r   rJ   r_   r`   ra   r#   r7   rM  )r   r  rf   rA   rC   r  r€  Úrepeate_regularizers           rD   rq  z#Optimizer.append_regularization_ops9  s7  € ð. ÐÜ×$Ñ$Ô&¬+¬-Û3‘��tØ×>Ñ>Ø˜4 ó�ð !×'Ñ'¨°Ð(9Õ:ð	  4ð2  Ðð' #(ÐÜ×%Ñ%Ð&6Õ7Û#7‘K�E˜4á/Ø!×-Ñ-Ð9Ø*Ð6à.2Ð+ÜŸ™ðIà,×4Ñ4Ó6ñ7ôð
 Ÿ™×,Ñ,×=Ñ=¸uÀd¸mÕLØ#'×#FÑ#FØ! 4¨ó$˜ð )×/Ñ/°¸Ð0AÔB÷	 MÐLñ $8÷ 8ð$  Ð÷ MÐLú÷ 8ð$  Ðús%   Á*A(DÃ'DÃ9
DÄDÄDÄD$c                 óü   — t        |«      }|j                  j                  j                  «       j	                  «       }|D �ch c]  }|j
                  du sŒ|j                  ’Œ }}|j                  |«       |S c c}w )NT)r   r#   r7   r$   r%   r»   r‚   rN  )r   re  r3   r�   rC   Úparam_no_trainables         rD   r^  zOptimizer._get_no_grad_setm  sv   € Ü+¨KÓ8ˆØ—Z‘Z×'Ñ'×4Ñ4Ó6×EÑEÓGˆ
á$.ó
Ù$.˜5°%×2EÑ2EÈÒ2MˆE�J‹J Jð 	ð 
ð 	×ÑÐ-Ô.àÐùò
s   ÁA9ÁA9c                 ó\  — g }| j                   �t        | j                   d   t        «      s0| j                   D ]   }|j                  rŒ|j	                  |«       Œ" n9| j
                  D ]*  }|d   D ]   }|j                  rŒ|j	                  |«       Œ" Œ, |D ]  }|j                  |«       Œ y)aª  
        Clear the gradients of all optimized parameters for model.

        If not, new gradient will accumulat on previous gradient.

        There are two method to clear grad: set_to_zero or delete grad.

        Args:
            set_to_zero (bool, optional): If set grads to zero or not, default is True.

        Returns:
            None

        Examples:
            .. code-block:: python

                >>> import paddle

                >>> a = paddle.arange(26, dtype="float32").reshape([2, 13])
                >>> linear = paddle.nn.Linear(13, 5)
                >>> # This can be any optimizer supported by dygraph.
                >>> adam = paddle.optimizer.Adam(learning_rate = 0.01,
                ...                             parameters = linear.parameters())
                >>> out = linear(a)
                >>> out.backward()
                >>> adam.step()
                >>> adam.clear_grad()

        Nr   rK   )r[   rR   rY   r»   r+   rt   Úclear_gradient)r   Úset_to_zeroÚ
param_listrQ  rƒ   s        rD   rq   zOptimizer.clear_gradx  s£   € ð> ˆ
Ø×ÑÐ'¬zØ× Ñ  Ñ#¤Tô0
ð ×)Ô)�Ø—“Ø×%Ñ% aÕ(ñ *ð  $×1Ô1�Ø$ XÔ.�AØŸ?›?Ø"×)Ñ)¨!Õ,ñ /ð  2ó
 ˆAØ×Ñ˜[Õ)ñ rF   c                 óÞ   — t        |t        t        j                  j                  f«      sJ d«       ‚|r|n| j
                  }| j                  ||||¬«      }| j                  |||¬«      }||fS )a  
        Add operations to minimize ``loss`` by updating ``parameters``.

        Args:
            loss (Tensor): A ``Tensor`` containing the value to minimize.
            startup_program (Program, optional): :ref:`api_paddle_static_Program` for
                initializing parameters in ``parameters``. The default value
                is None, at this time :ref:`api_paddle_static_default_startup_program` will be used.
            parameters (list, optional): List of ``Tensor`` or ``Tensor.name`` to update
                to minimize ``loss``. The default value is None, at this time all parameters
                will be updated.
            no_grad_set (set, optional): Set of ``Tensor``  or ``Tensor.name`` that don't need
                to be updated. The default value is None.

        Returns:
            tuple: tuple (optimize_ops, params_grads), A list of operators appended
            by minimize and a list of (param, grad) tensor pairs, param is
            ``Parameter``, grad is the gradient value corresponding to the parameter.
            In static graph mode, the returned tuple can be passed to ``fetch_list`` in ``Executor.run()`` to
            indicate program pruning. If so, the program will be pruned by ``feed`` and
            ``fetch_list`` before run, see details in ``Executor``.

        Examples:
            .. code-block:: python

                >>> import paddle
                >>> linear = paddle.nn.Linear(10, 10)
                >>> input = paddle.uniform(shape=[10, 10], min=-0.1, max=0.1)
                >>> out = linear(input)
                >>> loss = paddle.mean(out)

                >>> beta1 = paddle.to_tensor([0.9], dtype="float32")
                >>> beta2 = paddle.to_tensor([0.99], dtype="float32")

                >>> adam = paddle.optimizer.Adam(learning_rate=0.1,
                ...         parameters=linear.parameters(),
                ...         weight_decay=0.01)
                >>> loss.backward()
                >>> adam.minimize(loss)
                >>> adam.clear_grad()

        zThe loss should be an Tensor.)rØ   r�   r3   )rØ   rg  )rR   r   rS   rÈ   rË   r[   rj  rv  )r   re  rØ   r�   r3   r2   rg  rr  s           rD   ÚminimizezOptimizer.minimize§  sŽ   € ô\ Ø”8œVŸZ™Z×0Ñ0Ð1ô
ð 	+à*ó	+ð 
ñ (2™°t×7KÑ7Kˆà—}‘}ØØ+Ø%Ø#ð	 %ó 
ˆð ×+Ñ+Ø /Àð ,ó 
ˆð ˜\Ð)Ð)rF   c                 óè  ‡— t         j                  j                  «       j                  «       j	                  «       }t        | j                  d   t        «      rJ d«       ‚| j                  D �ch c]  }|j                  ’Œ c}Š|D �cg c]  }|j                  sŒ|‘Œ }}t        t        ˆfd„|«      «      }|D �cg c]  }||j                  f‘Œ }}| j                  |«      }yc c}w c c}w c c}w )zW
        In declarative mode, we forward `call step` to `call apply_gradients`
        r   zQOnly list of parameters is supported while using optimizer in @paddle.jit.static.c                 ó<   •— | j                   ‰v xr t        | d«      S )Nr  )r‚   r^   )ro  Úselected_paramss    €rD   rO   z-Optimizer._declarative_step.<locals>.<lambda>ö  s   ø€ ˜!Ÿ&™& OÐ3ÒJ¼ÀÀ6Ó8JÐJrF   N)rS   rÃ   r
   r$   r%   rR   r[   rY   r‚   r¾   rZ   Úfilterr  rs  )r   rK   rC   r�   rg  rr  rŽ  s         @rD   Ú_declarative_stepzOptimizer._declarative_stepè  sí   ø€ ô
 �M‰M×.Ñ.Ó0×=Ñ=Ó?×NÑNÓPð 	ô Ø× Ñ  Ñ#¤Tô
ð 	_à^ó	_ð 
ð 48×3GÒ3GÓHÑ3G¨%˜5Ÿ:›:Ð3GÑHˆÙ)/ÓC© °5·?³?’e¨ˆ
ÐCÜÜÛJØóó
ˆ
ñ :DÓD¹°˜ §
¡
Ò+¸ˆÐDØ×+Ñ+¨LÓ9‰ùò IùÚCùò Es   Á.C%ÂC*ÂC*Â<C/c           	      ó.  — t         j                  j                  j                  j                  «       r| j	                  «        yt        | j                  d   t        «      sjg }| j                  D ]C  }|j                  rŒ|j                  «       €Œ!|j                  «       }|j                  ||f«       ŒE | j                  dd|d¬«       yt        | j                  «      D ]­  \  }}t        d„ «      }|d   D ]F  }|j                  rŒ|j                  «       €Œ!|j                  «       }|d   j                  ||f«       ŒH |j                  |j                  «       D ��ci c]  \  }}|dk7  sŒ||“Œ c}}«       | j                  dd||¬«       Œ¯ yc c}}w )a©  
        Execute the optimizer and update parameters once.

        Returns:
            None

        Examples:
            .. code-block:: python

                >>> import paddle

                >>> a = paddle.arange(26, dtype="float32").reshape([2, 13])
                >>> linear = paddle.nn.Linear(13, 5)
                >>> # This can be any optimizer supported by dygraph.
                >>> adam = paddle.optimizer.Adam(learning_rate = 0.01,
                ...                         parameters = linear.parameters())
                >>> out = linear(a)
                >>> out.backward()
                >>> adam.step()
                >>> adam.clear_grad()
        Nr   )re  rØ   rg  rB  c                  ó   — g S rM   rN   rN   rF   rD   rO   z Optimizer.step.<locals>.<lambda>,  s   € ±2rF   rK   )rS   rÑ   ÚdygraphÚin_to_static_moder�  rR   rt   rY   r»   Ú
_grad_ivarr+   rv  r0   r   rN  rš   )r   rg  rC   Úgrad_varrE  rƒ   rŸ   r    s           rD   ÚstepzOptimizer.stepý  s�  € ô0 �;‰;×Ñ×#Ñ#×5Ñ5Ô7Ø×"Ñ"Ô$Øä˜$×,Ñ,¨QÑ/´Ô6ØˆLØ×+Ô+�Ø×&Ò&ØØ×#Ñ#Ó%Ñ1Ø$×/Ñ/Ó1�HØ ×'Ñ'¨°Ð(9Õ:ð ,ð × Ñ ØØ $Ø)Ø !ð	 !õ ô %.¨d×.@Ñ.@Ö$AÑ ��[Ü*©:Ó6�Ø(¨Ô2�EØ×*Ò*Ø Ø×'Ñ'Ó)Ñ5Ø#(×#3Ñ#3Ó#5˜Ø$ XÑ.×5Ñ5°u¸hÐ6GÕHð 3ð ×#Ñ#Ø&1×&7Ñ&7Ô&9ÔKÑ&9™d˜a ¸QÀ(»]�Q˜‘TÐ&9ÒKôð ×$Ñ$ØØ$(Ø!-Ø$'ð	 %õ ñ %Bùó Ls   ÅFÅ,Fc                 ó’  — |d   }t        |t        «      r|g|d<   n)t        |t        «      rt        d«      ‚t	        |«      |d<   | j
                  j                  «       D ]  \  }}|j                  ||«       Œ t        «       }| j                  D ]  }|j                  t        |d   «      «       Œ! |j                  t        |d   «      «      st        d«      ‚|d   D ]K  }|d   }t        |t        «      rt        |«      }	n|}	|	|_        |j                  dd«      |j                   d<   ŒM | j                  j#                  |«       y)zº
        Add a param group to parameter_list.

        Args:
            param_group (dict): The group of Tensors to be optimzed with
            different optimization options.
        rK   z`optimizer parameters should be in ordered collections,but received set, please use list instead.z7some parameters appear in more than one parameter grouprP   r€   rü   N)rR   r   r{   rV   rZ   rs   rš   Ú
setdefaultrt   rN  Ú
isdisjointÚ
ValueErrorrb   r   rJ   r”   rû   r+   )
r   rƒ   rK   rŸ   r    Ú	param_setÚgrouprC   rP   rf   s
             rD   ru   zOptimizer._add_param_group=  sI  € ð ˜XÑ&ˆÜ�fœiÔ(Ø%+ HˆK˜Ò!Ü˜¤Ô$Üð=óð ô
 %)¨£LˆK˜Ñ!ð ×&Ñ&×,Ñ,Ö.‰DˆAˆqØ×"Ñ" 1 aÕ(ð /ô “Eˆ	Ø×'Ô'ˆEØ×ÑœS  x¡Ó1Õ2ð (ð ×#Ñ#¤C¨°HÑ(=Ó$>Ô?ÜØIóð ð ! Ô*ˆEØ& ~Ñ6ˆLÜ˜,¬Ô.Ü!(¨Ó!6‘à!-�Ø .ˆEÔØ3>·?±?Ø ó4ˆE×Ñ Ò0ð +ð 	×Ñ×!Ñ! +Õ.rF   c                  ó   — y)zÙ
        Update the param group with new entry
        Args:
            parameters (dict): The extra group of Tensors to be optimzed with
            different optimization options. Only used in child class.
        NrN   )r   r�   s     rD   rK  zOptimizer._update_param_groupj  r  rF   c                  ó   — y)a©  
        All parameters used for optimizer (such as: parameters, master_weight, velocity_acc for momentum) calculations are grouped into a python list by data type (float16, float32).
        This function will be overridden in the corresponding optimizer file.

        Args:
            target_block: the block in which the loss tensor is present
            parameters: list of parameter tensors for the optimizer
        NrN   )r   r7  r�   rB  s       rD   rJ  zOptimizer._multi_tensor_inits  s   € ð 	rF   c                  ó   — y)zM
        For Multi Tensor, append optimize merged_operator to block.
        NrN   )r   r7  r  rB  s       rD   rL  z*Optimizer._append_optimize_multi_tensor_op  r  rF   c                 óØ  — t        |t        j                  j                  t        j                  f«      sJ d«       ‚t        |t        j                  j                  «      rP|t        j                  j                  j
                  k(  xs' |t        j                  j                  j                  k(  S |t        j                  j                  k(  xs |t        j                  j                  k(  S )z¼
        check the dtype is fp16 or the dtype is bf16
        :param dtype: instance of core.VarDesc.VarType
        :return: True if dtype is one of fp16 or bf16, False otherwise
        zIThe dtype should be an instance of core.VarDesc.VarType or core.DataType.)	rR   r   r	  r
  rÒ   ÚFP16ÚBF16ÚFLOAT16ÚUINT16)r   rj   s     rD   r  z Optimizer._is_dtype_fp16_or_bf16ˆ  sµ   € ô Ø”D—L‘L×(Ñ(¬$¯-©-Ð8ô
ð 	WàVó	Wð 
ô �eœTŸ\™\×1Ñ1Ô2àœŸ™×-Ñ-×2Ñ2Ñ2ò 6ØœDŸL™L×0Ñ0×5Ñ5Ñ5ðð œŸ™×.Ñ.Ñ.ÒO°%¼4¿=¹=×;OÑ;OÑ2OðrF   )NNNNrM   )Ng        NNN)r   )T)NNN)3rI  Ú
__module__Ú__qualname__Ú__doc__Úimperative_baseÚno_gradr„   r~   rŒ   rx   r•   r   Údygraph_onlyrž   r¥   r´   rä   rî   rò   rô   rÂ   rù   r  r  r  r  r  r  r(  r+  r1  r:  r"  rX  rZ  rj  rs  rv  r�  rq  r^  Únon_static_onlyrq   r‹  r�  r—  ru   rK  rJ  rL  r  rN   rF   rD   rH   rH   d   s2  „ ñGðR €_×ÑÓð ØØØòq*ó ðq*òf"ò
(ò
ò3ð ×Ññ%ó ð%ðN ×ÑñUNó ðUNòn$òzðx ×ÑñBó ðBðH ×Ññ+(ó ð+(òZY)óv
:ò
ò0ò6ò@.òò&òð" ØØØØóaòF4ò*5ò4ò ð 56ó}3ð@ 56óG3ðX ØØØókòZ*ðZ DEó4ól=ð@ 48ó2 óh	ð ×Ñò,*ó ð,*ð\ €_×ÑÓàGKò>*ó ð>*ò@:ð* €_×ÑÓØ×Ññ<ó ó ð<ò|+/òZð ×Ññ	ó ð	ð ×Ññó ðórF   rH   )NNNNN)0r_   ÚosÚcollectionsr   Únumpyr©   rS   Úpaddle.autogradrb  r©  r   Úpaddle._pir_opsr   r   Úpaddle.baser   Úpaddle.base.frameworkr   r	   r
   r   r   r   r   r   r   Úpaddle.regularizerr   rÑ   r   r   Úbase.backwardr   r   Úbase.frameworkr   Úbase.layer_helperr   rß   r   Ú__all__rê   Úenvironr”   ru  Ústatic_onlyrE   rH   rN   rF   rD   Ú<module>r»     s    ðó Û 	Ý #ã ã Ý )Ý ß 4Ý ÷
÷ 
õ 
õ 'ç )ß BÝ &Ý +Ý à
€á#&Ø‡J�J‡N�NÐ9¸1Ó=ó$Ð  ð
 ×Ñð ØØØØò,ó ð,÷^uò urF   