Ë
    •\;jÂu  ã                   ón   — d dl mZ d dlmZ d dlZddlmZ ddlmZ g Z	dd„Z
	 	 	 	 dd	„Z G d
„ de«      Zy)é    )Údefaultdict)ÚreduceNé   )Ú	frameworké   )Ú	Optimizerc                 óV  — |�|\  }}n| |k  r| |fn|| f\  }}||z   d||z
  z  | |z
  z  z
  }	|	dz  ||z  z
  }
|
dk\  rf|
j                  «       }| |k  r||| z
  ||z   |	z
  ||z
  d|z  z   z  z  z
  }n| | |z
  ||z   |	z
  ||z
  d|z  z   z  z  z
  }t        t        ||«      |«      S ||z   dz  S )a]  Cubic interpolation between (x1, f1, g1) and (x2, f2, g2).
        Use two points and their gradient to determine a cubic function and get the minimum point
        between them in the cubic curve.

    Reference:
        Jorge Nocedal, Stephen J. Wright, Numerical Optimization, Second Edition, 2006.
        pp59: formula 3.59

    Args:
        x1, f1, g1: point1's position, value and gradient.
        x2, f2, g2: point2's position, value and gradient.
        bounds: bounds of interpolation area

    Returns:
        min_pos: the minimum point between the specified points in the cubic curve.
    é   r   r   g       @)ÚsqrtÚminÚmax)Úx1Úf1Úg1Úx2Úf2Úg2ÚboundsÚ
xmin_boundÚ
xmax_boundÚd1Ú	d2_squareÚd2Úmin_poss                ú_G:\00. PROJECTS\API\Inventory\templateJSON\kerjaOCR\Lib\site-packages\paddle/optimizer/lbfgs.pyÚ_cubic_interpolater      sö   € ð$ ÐØ!'Ñˆ
‘Jà-/°2ªX " b¡¸BÀ¸8Ñˆ
�Jà	ˆb‰�1˜˜R™‘= B¨¡GÑ,Ñ	,€BØ�A‘˜˜R™‘€IØ�A‚~Ø�^‰^ÓˆØ�Š8Ø˜B ™G¨¨b©°2©¸"¸r¹'ÀAÈÁFÑ:JÑ(KÑLÑL‰Gà˜B ™G¨¨b©°2©¸"¸r¹'ÀAÈÁFÑ:JÑ(KÑLÑLˆGÜ”3�w 
Ó+¨ZÓ8Ð8à˜ZÑ'¨3Ñ.Ð.ó    c           
      óh  — |j                  «       j                  «       }|j                  «       } | |||«      \  }}d}t        j                  ||«      }t        j
                  d|j                  ¬«      |||f\  }}}}d}d}||
k  rí||||z  |z  z   kD  s
|dkD  r$||k\  r||g}||g}||j                  «       g}||g}n¶t        j                   |«      | |z  k  r|g}|g}|g}d}nŽ|dk\  r||g}||g}||j                  «       g}||g}nj|d||z
  z  z   }|dz  }|}t        ||||||||f¬«      }|}|}|j                  «       }|} | |||«      \  }}|dz  }|j	                  |«      }|dz  }||
k  rŒí||
k(  rd|g}||g}||g}d}d   |d	   k  rd
nd\  }}|�s||
k  �rýt        j                   d   |d   z
  «      |z  |	k  r�n×t        |d   |d   d   |d   |d   |d   «      }dt        |«      t        |«      z
  z  } t        t        |«      |z
  |t        |«      z
  «      | k  r„|s|t        |«      k\  s|t        |«      k  rct        j                   |t        |«      z
  «      t        j                   |t        |«      z
  «      k  rt        |«      | z
  }nt        |«      | z   }d}nd}nd} | |||«      \  }}|dz  }|j	                  |«      }|dz  }||||z  |z  z   kD  s|||   k\  r5|||<   |||<   |j                  «       |<   |||<   |d   |d   k  rd
nd\  }}nrt        j                   |«      | |z  k  rd}n1|||   ||   z
  z  dk\  r ||   ||<   ||   ||<   |   ||<   ||   ||<   |||<   |||<   |j                  «       |<   |||<   |s||
k  r�Œý|   }||   }|   }||||fS )ag  Implements of line search algorithm that satisfies the strong Wolfe conditions using double zoom.

    Reference:
        Jorge Nocedal, Stephen J. Wright, Numerical Optimization, Second Edition, 2006.
        pp60: Algorithm 3.5 (Line Search Algorithm).

    Args:
        obj_func: the objective function to minimize. ```` accepts a multivariate input and returns a scalar.
        xk (Tensor): the starting point of the iterates.
        alpha (Scalar): the initial step size.
        d (Tensor): search direction.
        loss (scalar): the initial loss
        grad (Tensor): the initial grad
        c1 (Scalar): parameter for sufficient decrease condition.
        c2 (Scalar): parameter for curvature condition.
        tolerance_change (Scalar): terminates if the change of function value/position/parameter between
            two iterations is smaller than this value.
        max_ls(int): max iteration of line search.
        alpha_max (float): max step length.

    Returns:
        loss_new (Scaler): loss of obj_func at final alpha.
        grad_new, (Tensor): derivative of obj_func at final alpha.
        alpha(Tensor): optimal step length, or 0. if the line search algorithm did not converge.
        ls_func_evals (Scaler): number of objective function called in line search process.

    Following summarizes the essentials of the strong Wolfe line search algorithm.
    Some notations used in the description:

        - `func` denotes the objective function.
        - `obi_func` is a function of step size alpha, restricting `obj_func` on a line.

            obi_func = func(xk + alpha * d),
            where xk is the position of k'th iterate, d is the line search direction(decent direction),
            and a is the step size.
        - alpha : substitute of alpha
        - a1 is alpha of last iteration, which is alpha_(i-1).
        - a2 is alpha of current iteration, which is alpha_i.
        - a_lo is alpha in left position when calls zoom, which is alpha_low.
        - a_hi is alpha in right position when calls zoom, which is alpha_high.

    Line Search Algorithm:
        repeat
            Compute obi_func(a2) and derphi(a2).
            1. If obi_func(a2) > obi_func(0) + c_1 * a2 * obi_func'(0) or [obi_func(a2) >= obi_func(a1) and i > 1],
                alpha= zoom(a1, a2) and stop;

            2. If |obi_func'(a2)| <= -c_2 * obi_func'(0),
                alpha= a2 and stop;

            3. If obi_func'(a2) >= 0,
                alpha= zoom(a2, a1) and stop;

            a1 = a2
            a2 = min(2 * a2, a2)
            i = i + 1
        end(repeat)

    zoom(a_lo, a_hi) Algorithm:
        repeat
            aj = cubic_interpolation(a_lo, a_hi)
            Compute obi_func(aj) and derphi(aj).
            1. If obi_func(aj) > obi_func(0) + c_1 * aj * obi_func'(0) or obi_func(aj) >= obi_func(a_lo),
                then a_hi <- aj;
            2.
                2.1. If |obi_func'(aj)| <= -c_2 * obi_func'(0), then alpha= a2 and stop;

                2.2. If obi_func'(aj) * (a2 - a1) >= 0, then a_hi = a_lo

                a_lo = aj;
        end(repeat)

    reference: https://github.com/pytorch/pytorch
    r   r   ©ÚdtypeFTg{®Gáz„?é
   )r   éÿÿÿÿ)r   r   )r   r   gš™™™™™¹?)	Úabsr   ÚcloneÚpaddleÚdotÚ	to_tensorr    r   r   )!Úobj_funcÚxkÚalphaÚdÚlossÚgradÚgtdÚc1Úc2Útolerance_changeÚmax_lsÚd_normÚloss_newÚgrad_newÚls_func_evalsÚgtd_newÚt_prevÚf_prevÚg_prevÚgtd_prevÚdoneÚls_iterÚbracketÚ	bracket_fÚ	bracket_gÚbracket_gtdÚmin_stepÚmax_stepÚtmpÚinsuf_progressÚlow_posÚhigh_posÚepss!                                    r   Ú_strong_wolferI   >   sÛ  € ðp �U‰U‹W�[‰[‹]€FØ�:‰:‹<€Dá! " e¨QÓ/Ñ€HˆhØ€MÜ�j‰j˜ 1Ó%€Gô 	×Ñ˜ $§*¡*Ô-ØØØð	(Ñ$€FˆF�F˜Hð €DØ€GØ
�FÒ
à�t˜b 5™j¨3Ñ.Ñ.Ò/Ø�aŠK˜H¨Ò.à˜u�oˆGØ Ð*ˆIØ §¡Ó!1Ð2ˆIØ# WÐ-ˆKØä�:‰:�gÓ 2 #¨¡)Ò+Ø�gˆGØ!˜
ˆIØ!˜
ˆIØˆDØà�aŠ<Ø˜u�oˆGØ Ð*ˆIØ §¡Ó!1Ð2ˆIØ# WÐ-ˆKØð ˜4 5¨6¡>Ñ2Ñ2ˆØ˜2‘:ˆØˆÜ"ØØØØØØØ˜hÐ'ô
ˆð ˆØˆØ—‘Ó!ˆØˆá% b¨%°Ó3Ñˆ�(Ø˜ÑˆØ—,‘,˜q“/ˆØ�1‰ˆða �FÓ
ðf �&ÒØ�e�*ˆØ˜8Ð$ˆ	Ø˜8Ð$ˆ	ð
 €Nà"+¨A¡,°)¸B±-Ò"?™ÀVÑ€GˆXÚ�w Ó'ä�:‰:�g˜a‘j 7¨1¡:Ñ-Ó.°Ñ7Ð:JÒJÙô #Ø�A‰JØ�a‰LØ˜‰NØ�A‰JØ�a‰LØ˜‰Nó
ˆð" ”S˜“\¤C¨£LÑ0Ñ1ˆÜŒs�7‹|˜eÑ# U¬S°«\Ñ%9Ó:¸SÒ@á ¬#¨g«,Ò!6¸%Ä3ÀwÃ<Ò:Oä—:‘:˜e¤c¨'£lÑ2Ó3´f·j±jØœC ›LÑ(ó7ò ô   ›L¨3Ñ.‘Eä ›L¨3Ñ.�EØ!&‘à!%‘à"ˆNá% b¨%°Ó3Ñˆ�(Ø˜ÑˆØ—,‘,˜q“/ˆØ�1‰ˆð ˜˜r E™z¨CÑ/Ñ/Ò0Ø˜9 WÑ-Ò-ð !&ˆG�HÑØ"*ˆI�hÑØ"*§.¡.Ó"2ˆI�hÑØ$+ˆK˜Ñ!à# A™,¨)°A©,Ò6‘¸Fñ ˆG‘Xô �z‰z˜'Ó" r c¨C¡iÒ/à‘Ø˜G HÑ-°¸Ñ0@Ñ@ÑAÀQÒFà$+¨GÑ$4�˜Ñ!Ø&/°Ñ&8�	˜(Ñ#Ø&/°Ñ&8�	˜(Ñ#Ø(3°GÑ(<�˜HÑ%ð  %ˆG�GÑØ!)ˆI�gÑØ!)§¡Ó!1ˆI�gÑØ#*ˆK˜Ñ ñQ �w Ô'ðV �GÑ€EØ˜Ñ!€HØ˜Ñ!€HØ�X˜u mÐ3Ð3r   c                   ó’   ‡ — e Zd ZdZ	 	 	 	 	 	 	 	 	 	 	 dˆ fd„	Zd„ Zd„ Zd„ Zd„ Zd„ Z	d„ Z
d	„ Zej                  d
„ «       Z	 dd„Zˆ xZS )ÚLBFGSa  
    The L-BFGS is a quasi-Newton method for solving an unconstrained optimization problem over a differentiable function.
    Closely related is the Newton method for minimization. Consider the iterate update formula:

    .. math::
        x_{k+1} = x_{k} + H_k \nabla{f_k}

    If :math:`H_k` is the inverse Hessian of :math:`f` at :math:`x_k`, then it's the Newton method.
    If :math:`H_k` is symmetric and positive definite, used as an approximation of the inverse Hessian, then
    it's a quasi-Newton. In practice, the approximated Hessians are obtained
    by only using the gradients, over either whole or part of the search
    history, the former is BFGS, the latter is L-BFGS.

    Reference:
        Jorge Nocedal, Stephen J. Wright, Numerical Optimization, Second Edition, 2006. pp179: Algorithm 7.5 (L-BFGS).

    Args:
        learning_rate (float, optional): learning rate .The default value is 1.
        max_iter (int, optional): maximal number of iterations per optimization step.
            The default value is 20.
        max_eval (int, optional): maximal number of function evaluations per optimization
            step. The default value is max_iter * 1.25.
        tolerance_grad (float, optional): termination tolerance on first order optimality
            The default value is 1e-5.
        tolerance_change (float, optional): termination tolerance on function
            value/parameter changes. The default value is 1e-9.
        history_size (int, optional): update history size. The default value is 100.
        line_search_fn (string, optional): either 'strong_wolfe' or None. The default value is strong_wolfe.
        parameters (list|tuple, optional): List/Tuple of ``Tensor`` names to update to minimize ``loss``. \
            This parameter is required in dygraph mode. The default value is None.
        weight_decay (float|WeightDecayRegularizer, optional): The strategy of regularization. \
            It canbe a float value as coeff of L2 regularization or \
            :ref:`api_paddle_regularizer_L1Decay`, :ref:`api_paddle_regularizer_L2Decay`.
            If a parameter has set regularizer using :ref:`api_paddle_ParamAttr` already, \
            the regularization setting here in optimizer will be ignored for this parameter. \
            Otherwise, the regularization setting here in optimizer will take effect. \
            Default None, meaning there is no regularization.
        grad_clip (GradientClipBase, optional): Gradient cliping strategy, it's an instance of \
            some derived class of ``GradientClipBase`` . There are three cliping strategies \
            ( :ref:`api_paddle_nn_ClipGradByGlobalNorm` , :ref:`api_paddle_nn_ClipGradByNorm` , \
            :ref:`api_paddle_nn_ClipGradByValue` ). Default None, meaning there is no gradient clipping.
        name (str, optional): Normally there is no need for user to set this property.
            For more information, please refer to :ref:`api_guide_Name`.
            The default value is None.

    Return:
        loss (Tensor): the final loss of closure.

    Examples:
        .. code-block:: python

            >>> import paddle
            >>> import numpy as np

            >>> paddle.disable_static()
            >>> np.random.seed(0)
            >>> np_w = np.random.rand(1).astype(np.float32)
            >>> np_x = np.random.rand(1).astype(np.float32)

            >>> inputs = [np.random.rand(1).astype(np.float32) for i in range(10)]
            >>> # y = 2x
            >>> targets = [2 * x for x in inputs]

            >>> class Net(paddle.nn.Layer):
            ...     def __init__(self):
            ...         super().__init__()
            ...         w = paddle.to_tensor(np_w)
            ...         self.w = paddle.create_parameter(shape=w.shape, dtype=w.dtype, default_initializer=paddle.nn.initializer.Assign(w))
            ...
            ...     def forward(self, x):
            ...         return self.w * x
            ...
            >>> net = Net()
            >>> opt = paddle.optimizer.LBFGS(learning_rate=1, max_iter=1, max_eval=None, tolerance_grad=1e-07, tolerance_change=1e-09, history_size=100, line_search_fn='strong_wolfe', parameters=net.parameters())
            >>> def train_step(inputs, targets):
            ...     def closure():
            ...         outputs = net(inputs)
            ...         loss = paddle.nn.functional.mse_loss(outputs, targets)
            ...         print('loss: ', loss.item())
            ...         opt.clear_grad()
            ...         loss.backward()
            ...         return loss
            ...     opt.step(closure)
            ...
            >>> for input, target in zip(inputs, targets):
            ...     input = paddle.to_tensor(input)
            ...     target = paddle.to_tensor(target)
            ...     train_step(input, target)
    c                 óö  •— |€|dz  dz  }|| _         || _        || _        || _        || _        || _        || _        t        |t        j                  «      rt        dt        |«      z   «      ‚t        t        «      | _        t        ‰| �A  d||	|
|¬«       t        | j"                  d   t        «      s| j"                  | _        d | _        y t'        | j(                  «      D ]  \  }}|d   | _        Œ d | _        y )Né   é   z^parameters argument given to the optimizer should be an iterable of Tensors or dicts, but got ç      ð?)Úlearning_rateÚ
parametersÚweight_decayÚ	grad_clipÚnamer   Úparams)rP   Úmax_iterÚmax_evalÚtolerance_gradr1   Úhistory_sizeÚline_search_fnÚ
isinstancer%   ÚTensorÚ	TypeErrorÚtyper   ÚdictÚstateÚsuperÚ__init__Ú_parameter_listÚ_paramsÚ	enumerateÚ_param_groupsÚ_numel_cache)ÚselfrP   rV   rW   rX   r1   rY   rZ   rQ   rR   rS   rT   ÚidxÚparam_groupÚ	__class__s                 €r   rb   zLBFGS.__init__�  s  ø€ ð ÐØ !‘| qÑ(ˆHà*ˆÔØ ˆŒØ ˆŒØ,ˆÔØ 0ˆÔØ(ˆÔØ,ˆÔä�j¤&§-¡-Ô0Üð<Ü>BÀ:Ó>NñOóð ô
 !¤Ó&ˆŒ
ä‰ÑØØ!Ø%ØØð 	ô 	
ô ˜$×.Ñ.¨qÑ1´4Ô8Ø×/Ñ/ˆDŒLð
 !ˆÕô %.¨d×.@Ñ.@Ö$AÑ ��[Ø*¨8Ñ4�•ð %Bð !ˆÕr   c                 óx   — i }| j                   j                  «       D ]  \  }}|j                  ||i«       Œ d|iS )a„  Returns the state of the optimizer as a :class:`dict`.

        Return:
            state, a dict holding current optimization state. Its content
            differs between optimizer classes.

        Examples:
            .. code-block:: python

                >>> import paddle

                >>> paddle.disable_static()

                >>> net = paddle.nn.Linear(10, 10)
                >>> opt = paddle.optimizer.LBFGS(
                ...     learning_rate=1,
                ...     max_iter=1,
                ...     max_eval=None,
                ...     tolerance_grad=1e-07,
                ...     tolerance_change=1e-09,
                ...     history_size=100,
                ...     line_search_fn='strong_wolfe',
                ...     parameters=net.parameters(),
                >>> )

                >>> def train_step(inputs, targets):
                ...     def closure():
                ...         outputs = net(inputs)
                ...         loss = paddle.nn.functional.mse_loss(outputs, targets)
                ...         opt.clear_grad()
                ...         loss.backward()
                ...         return loss
                ...
                ...     opt.step(closure)
                ...
                >>> inputs = paddle.rand([10, 10], dtype="float32")
                >>> targets = paddle.to_tensor([2 * x for x in inputs])

                >>> n_iter = 0
                >>> while n_iter < 20:
                ...     loss = train_step(inputs, targets)
                ...     n_iter = opt.state_dict()["state"]["func_evals"]
                ...     print("n_iter:", n_iter)
                ...
        r`   )r`   ÚitemsÚupdate)rh   Úpacked_stateÚkÚvs       r   Ú
state_dictzLBFGS.state_dictÁ  sD   € ð^ ˆØ—J‘J×$Ñ$Ö&‰DˆAˆqØ×Ñ  A Õ'ð 'ð ˜Ð&Ð&r   c                 ól   — | j                   €t        d„ | j                  d«      | _         | j                   S )Nc                 ó(   — | |j                  «       z   S ©N)Únumel)ÚtotalÚps     r   Ú<lambda>zLBFGS._numel.<locals>.<lambda>ú  s   €  ¨¯©«Ò!2r   r   )rg   r   rd   )rh   s    r   Ú_numelzLBFGS._numelö  s4   € à×ÑÐ$Ü &Ù2°D·L±LÀ!ó!ˆDÔð × Ñ Ð r   c                 ó  — g }| j                   D ]a  }|j                  €&t        j                  |«      j	                  dg«      }n|j                  j	                  dg«      }|j                  |«       Œc t        j                  |d¬«      S )Nr"   r   )Úaxis)rd   r-   r%   Ú
zeros_likeÚreshapeÚappendÚconcat)rh   Úviewsrx   Úviews       r   Ú_gather_flat_gradzLBFGS._gather_flat_gradÿ  sn   € ØˆØ—”ˆAØ�v‰vˆ~Ü×(Ñ(¨Ó+×3Ñ3°R°DÓ9‘à—v‘v—~‘~ r dÓ+�Ø�L‰L˜Õð ô �}‰}˜U¨Ô+Ð+r   c           	      ó<  — d}| j                   D ]v  }|j                  g k7  rt        d„ |j                  «      nd}t        j                  |j                  ||||z    j                  |j                  «      |z  «      |«      }||z  }Œx || j                  «       k(  sJ ‚y )Nr   c                 ó   — | |z  S ru   © )ÚxÚys     r   ry   z!LBFGS._add_grad.<locals>.<lambda>  s   € ¨¨Aªr   r   )rd   Úshaper   r%   ÚassignÚaddr~   rz   )rh   r*   Ú	directionÚoffsetrx   rv   s         r   Ú	_add_gradzLBFGS._add_grad
  s”   € ØˆØ—”ˆAØ;<¿7¹7Àbº=”FÑ-¨q¯w©wÔ7ÈaˆEÜ—‘Ø—‘Ø˜f v°¡~Ð6×>Ñ>¸q¿w¹wÓGÈ%ÑOóð ó	ˆAð �e‰O‰Fð ð ˜Ÿ™›Ò&Ð&Ñ&r   c                 ó\   — | j                   D �cg c]  }|j                  «       ‘Œ c}S c c}w ru   )rd   r$   )rh   rx   s     r   Ú_clone_paramzLBFGS._clone_param  s$   € Ø#'§<¢<Ó0¡<˜a�—‘•	 <Ñ0Ð0ùÒ0s   �)c                 ól   — t        | j                  |«      D ]  \  }}t        j                  ||«       Œ y ru   )Úziprd   r%   rŠ   )rh   Úparams_datarx   Úpdatas       r   Ú
_set_paramzLBFGS._set_param  s)   € Ü˜DŸL™L¨+Ö6‰HˆAˆuÜ�M‰M˜% Õ#ñ 7r   c                 ó�   — | j                  ||«       t         |«       «      }| j                  «       }| j                  |«       ||fS ru   )rŽ   Úfloatrƒ   r•   )rh   Úclosurer‡   r*   r+   r,   Ú	flat_grads          r   Ú_directional_evaluatezLBFGS._directional_evaluate  s@   € Ø�‰�u˜aÔ Ü‘W“YÓˆØ×*Ñ*Ó,ˆ	Ø�‰˜ÔØ�YˆÐr   c           
      óÚ  ‡ ‡— t        j                  «       5   t        j                  «       ‰«      Š‰ j                  }‰ j                  }‰ j
                  }‰ j                  }‰ j                  }‰ j                  }‰ j                  }‰ j                  }	|	j                  dd«       |	j                  dd«        ‰«       }
t        |
«      }d}|	dxx   dz  cc<   ‰ j                  «       }|j                  «       j                  «       |k  }|r|
cddd«       S |	j!                  d«      }|	j!                  d«      }|	j!                  d«      }|	j!                  d	«      }|	j!                  d
«      }|	j!                  d«      }|	j!                  d«      }|	j!                  d«      }d}||k  �rý|dz  }|	dxx   dz  cc<   |	d   dk(  r9|j#                  «       }g }g }g }t        j$                  d|
j&                  ¬«      }�nã|j)                  |«      }|j+                  t        j$                  ||j&                  ¬«      «      }|j-                  |«      }|dkD  r‹t/        |«      |k(  r3|j1                  d«       |j1                  d«       |j1                  d«       |j3                  |«       |j3                  |«       |j3                  d|z  «       ||j-                  |«      z  }t/        |«      }d|	vr	dg|z  |	d<   |	d   }|j#                  «       }t5        |dz
  dd«      D ]N  }||   j-                  |«      ||   z  ||<   t        j6                  |j9                  ||   ||    z  «      |«       ŒP t        j*                  ||«      x}}t5        |«      D ]M  }||   j-                  |«      ||   z  } t        j6                  |j9                  ||   ||   | z
  z  «      |«       ŒO |€|j;                  «       }nt        j6                  ||«       |}|	d   dk(  r/t=        dd|j                  «       j?                  «       z  «      |z  }n|}|j-                  |«      }!|!| kD  r�nJd}"|�p|dk7  rtA        d«      ‚‰ jC                  «       }#ˆˆ fd„}$tE        |$|#|||||!«      \  }}}}"‰ jG                  ||«       |j                  «       j                  «       |k  }nw‰ jG                  ||«       ||k7  r`t        j                  «       5  t         ‰«       «      }ddd«       ‰ j                  «       }|j                  «       j                  «       |k  }d}"||"z  }|	dxx   |"z  cc<   |rnJ||z  j                  «       j                  «       |k  rn%t        ||z
  «      |k  rn||k\  rn||k(  rn||k  r�Œý||	d<   ||	d<   ||	d<   ||	d	<   ||	d
<   ||	d<   ||	d<   ||	d<   ddd«       |
S # 1 sw Y   ŒÍxY w# 1 sw Y   
S xY w)a  Performs a single optimization step.

        Args:
            closure (callable): A closure that reevaluates the model
            and returns the loss.

        Examples:
            .. code-block:: python

                >>> import paddle

                >>> paddle.disable_static()

                >>> inputs = paddle.rand([10, 10], dtype="float32")
                >>> targets = paddle.to_tensor([2 * x for x in inputs])

                >>> net = paddle.nn.Linear(10, 10)
                >>> opt = paddle.optimizer.LBFGS(
                ...     learning_rate=1,
                ...     max_iter=1,
                ...     max_eval=None,
                ...     tolerance_grad=1e-07,
                ...     tolerance_change=1e-09,
                ...     history_size=100,
                ...     line_search_fn='strong_wolfe',
                ...     parameters=net.parameters(),
                >>> )

                >>> def closure():
                ...     outputs = net(inputs)
                ...     loss = paddle.nn.functional.mse_loss(outputs, targets)
                ...     print("loss:", loss.item())
                ...     opt.clear_grad()
                ...     loss.backward()
                ...     return loss
                ...
                >>> opt.step(closure)
        Ú
func_evalsr   Ún_iterr   Nr+   r*   Úold_ykÚold_skÚroÚH_diagÚprev_flat_gradÚ	prev_lossrO   r   g»½×Ùß|Û=Úalr"   Ústrong_wolfez only 'strong_wolfe' is supportedc                 ó,   •— ‰j                  ‰| ||«      S ru   )rš   )r‡   r*   r+   r˜   rh   s      €€r   r(   zLBFGS.step.<locals>.obj_funcÐ  s   ø€ Ø#'×#=Ñ#=Ø '¨¨E°1ó$ð r   )$r%   Úno_gradÚenable_gradrP   rV   rW   rX   r1   rZ   rY   r`   Ú
setdefaultr—   rƒ   r#   r   ÚgetÚnegr'   r    ÚsubtractÚmultiplyr&   ÚlenÚpopr   ÚrangerŠ   r‹   r$   r   ÚsumÚRuntimeErrorr�   rI   rŽ   )%rh   r˜   rP   rV   rW   rX   r1   rZ   rY   r`   Ú	orig_lossr,   Úcurrent_evalsr™   Úopt_condr+   r*   rž   rŸ   r    r¡   r¢   r£   r�   rˆ   ÚsÚysÚnum_oldr¤   ÚqÚiÚrÚbe_ir.   r6   Úx_initr(   s%   ``                                   r   Ústepz
LBFGS.step%  sÖ  ù€ ôR �^‰^Õà*”f×(Ñ(Ó*¨7Ó3ˆGà ×.Ñ.ˆMØ—}‘}ˆHØ—}‘}ˆHØ!×0Ñ0ˆNØ#×4Ñ4ÐØ!×0Ñ0ˆNØ×,Ñ,ˆLØ—J‘JˆEØ×Ñ˜\¨1Ô-Ø×Ñ˜X qÔ)ñ  ›	ˆIÜ˜Ó#ˆDàˆMØ�,Ó 1Ñ$Óà×.Ñ.Ó0ˆIØ —}‘}“×*Ñ*Ó,°Ñ>ˆHñ Ø ÷7 Ñð< —	‘	˜#“ˆAØ—I‘I˜gÓ&ˆEØ—Y‘Y˜xÓ(ˆFØ—Y‘Y˜xÓ(ˆFØ—‘˜4“ˆBØ—Y‘Y˜xÓ(ˆFØ"ŸY™YÐ'7Ó8ˆNØŸ	™	 +Ó.ˆIàˆFà˜8Ó#à˜!‘�Ø�h“ 1Ñ$“ð
 ˜‘? aÒ'Ø!Ÿ™›�AØ�FØ�FØ�BÜ#×-Ñ-¨c¸¿¹ÔI’Fð "×*Ñ*¨>Ó:�AØŸ
™
¤6×#3Ñ#3°EÀÇÁÔ#IÓJ�AØŸ™˜q›�BØ˜E’zä˜v›;¨,Ò6à"ŸJ™J qœMØ"ŸJ™J qœMØŸF™F 1œIð Ÿ™ aÔ(ØŸ™ aÔ(ØŸ	™	 #¨¡(Ô+ð "$ a§e¡e¨A£h¡˜ô " &›k�Gà 5Ñ(Ø'+ f¨|Ñ&;˜˜d™Ø˜t™�Bð "Ÿ™›�AÜ" 7¨Q¡;°°BÖ7˜Ø & q¡	§¡¨aÓ 0°2°a±5Ñ 8˜˜1™ÜŸ™ a§e¡e¨F°1©I¸"¸Q¹%¸Ñ,@Ó&AÀ1ÕEð 8ô #ŸO™O¨A¨vÓ6Ð6�A˜Ü" 7ž^˜Ø% a™yŸ}™}¨QÓ/°"°Q±%Ñ7˜ÜŸ™ a§e¡e¨F°1©I¸¸A¹À¹Ñ,FÓ&GÈÕKð ,ð "Ð)Ø%.§_¡_Ó%6‘Nä—M‘M )¨^Ô<Ø �	ð ˜‘? aÒ'ä˜C  y§}¡}£×':Ñ':Ó'<Ñ!<Ó=ÀÑMñ ð *�Eð  —m‘m AÓ&�ð Ð*Ð*Ò*Ùð !"�Ø!Ð-à%¨Ò7Ü*Ð+MÓNÐNà!%×!2Ñ!2Ó!4˜õô
 ANØ$ f¨e°Q¸¸iÈóAÑ=˜˜i¨°ð —N‘N 5¨!Ô,Ø(Ÿ}™}›×2Ñ2Ó4¸ÑF‘Hð —N‘N 5¨!Ô,Ø Ò)Ü#×/Ñ/Õ1Ü#(©«Ó#3˜D÷ 2à$(×$:Ñ$:Ó$<˜	Ø#,§=¡=£?×#6Ñ#6Ó#8¸NÑ#J˜Ø()˜ð  Ñ.�Ø�lÓ# }Ñ4Ó#ñ Øð ˜‘I—?‘?Ó$×(Ñ(Ó*Ð.>Ò>Øä�t˜iÑ'Ó(Ð+;Ò;Øð ! HÒ,Øà˜XÒ%ØðC ˜8Ô#ðF ˆE�#‰JØ"ˆE�'‰NØ$ˆE�(‰OØ$ˆE�(‰OØˆE�$‰KØ$ˆE�(‰OØ&4ˆEÐ"Ñ#Ø!*ˆE�+Ñ÷g ðj Ð÷K 2Ð1ú÷a ðj Ðús2   —C4W ÄO"W Ó7WÔBW Ö"(W ×W	×W × W*c                 ó   — t        d«      ‚)z}Empty method. LBFGS optimizer does not use this way to minimize ``loss``. Please refer 'Examples' of LBFGS() above for usage.zeLBFGS optimizer does not use this way to minimize loss. Please refer 'Examples' of LBFGS() for usage.)ÚNotImplementedError)rh   r,   Ústartup_programrQ   Úno_grad_sets        r   ÚminimizezLBFGS.minimize  s   € ô "Øsó
ð 	
r   )rO   é   NgH¯¼šò×z>ç•Ö&è.>éd   NNNNN)NNN)Ú__name__Ú
__module__Ú__qualname__Ú__doc__rb   rr   rz   rƒ   rŽ   r�   r•   rš   r   Únon_static_onlyr¾   rÃ   Ú__classcell__)rk   s   @r   rK   rK   5  s€   ø„ ñXðx ØØØØØØØØØØõ/!òb3'òj!ò,ò'ò1ò$òð ×Ññ]ó ð]ð@ HL÷
r   rK   ru   )g-Cëâ6?gÍÌÌÌÌÌì?rÅ   é   )Úcollectionsr   Ú	functoolsr   r%   Úbaser   Ú	optimizerr   Ú__all__r   rI   rK   r†   r   r   Ú<module>rÓ      sD   ðõ $Ý ã å Ý  à
€ó!/ðX Ø
ØØót4ônV
ˆIõ V
r   