Ë
    •\;jÒ3  ã                   óž   — d dl Z d dlmZ ddlmZmZmZ ddlmZ ddl	m
Z
mZmZ ddlmZ g Z G d	„ d
e«      Z G d„ de«      Z G d„ de«      Zy)é    N)Ú_C_opsé   )ÚcoreÚ	frameworkÚunique_name)Úcheck_variable_and_dtype)Ú_current_expected_placeÚin_dygraph_modeÚin_pir_modeé   )ÚInitializerc                   ó,   ‡ — e Zd ZdZdˆ fd„	Zdd„Zˆ xZS )ÚXavierInitializera·  
    This class implements the Xavier weight initializer from the paper
    `Understanding the difficulty of training deep feedforward neural
    networks <http://proceedings.mlr.press/v9/glorot10a/glorot10a.pdf>`_
    by Xavier Glorot and Yoshua Bengio.

    This initializer is designed to keep the scale of the gradients
    approximately same in all the layers. In case of Uniform distribution,
    the range is [-x, x], where

    .. math::

        x = \sqrt{\\frac{6.0}{fan\_in + fan\_out}}

    In case of Normal distribution, the mean is 0 and the standard deviation
    is

    .. math::

        \sqrt{\\frac{2.0}{fan\_in + fan\_out}}


    Args:
        uniform (bool, optional): whether to use uniform ,if False use normal distribution. Default is True.
        fan_in (float, optional): fan_in for Xavier initialization. If None, it is
                inferred from the variable. Default is None.
        fan_out (float, optional): fan_out for Xavier initialization. If None, it is
                 inferred from the variable. Default is None.
        seed (int, optional): Random seed. Default is 0.

    Note:
        It is recommended to set fan_in and fan_out to None for most cases.

    c                 ój   •— |€J ‚|€J ‚t         ‰| �  «        || _        || _        || _        || _        y ©N)ÚsuperÚ__init__Ú_uniformÚ_fan_inÚ_fan_outÚ_seed)ÚselfÚuniformÚfan_inÚfan_outÚseedÚ	__class__s        €úeG:\00. PROJECTS\API\Inventory\templateJSON\kerjaOCR\Lib\site-packages\paddle/nn/initializer/xavier.pyr   zXavierInitializer.__init__C   sB   ø€ ØÐ"Ð"Ð"ØÐÐÐÜ‰ÑÔØˆŒØˆŒØˆŒØˆ�
ó    c                 ó6  — ddl }| j                  |«      }t        |t        j                  |j
                  j                  f«      sJ ‚t        ||j
                  j                  j                  «      st        |dg d¢d«       | j                  |«      \  }}| j                  €|n| j                  }| j                  €|n| j                  }| j                  dk(  r|j                  j                  | _        |j                  t        j                   j"                  j$                  k(  s=|j                  t        j                   j"                  j&                  k(  r¢| j(                  s–t        j                   j"                  j*                  }|j-                  t/        j0                  dj3                  d|j4                  dg«      «      |j6                  |t        j                   j"                  j8                  d¬	«      }	nw|j                  t        j:                  j<                  t        j:                  j>                  fv r)| j(                  st        j:                  j@                  }|}	n|j                  }|}	tC        «       �rv| j(                  r\tE        jF                  d
tI        ||z   «      z  «      }
tK        jL                  |	j6                  ||
 |
| j                  tO        «       «      }	n\tE        jF                  dtI        ||z   «      z  «      }tO        «       }tK        jP                  |	j6                  d|| j                  ||«      }	|j                  t        j                   j"                  j$                  k(  s=|j                  t        j                   j"                  j&                  k(  r>| j(                  s2tK        jR                  |	|j                  «      }|jU                  |«       y|	jU                  |«       ytW        «       �r6| j(                  rbtE        jF                  d
tI        ||z   «      z  «      }
|jX                  jM                  |	j6                  ||
 |
| j                  tO        «       «      }	nZtE        jF                  dtI        ||z   «      z  «      }tK        jP                  |	j6                  d|| j                  |tO        «       «      }	|j                  t        j:                  j<                  t        j:                  j>                  fv r,| j(                  s tK        jR                  |	|j                  «      S |	S | j(                  rXtE        jF                  d
tI        ||z   «      z  «      }
|j[                  di d|	i|	j6                  ||
 |
| j                  dœd¬«      }n_tE        jF                  dtI        ||z   «      z  «      }|j[                  dd|	i|	j6                  |	j                  d|| j                  dœd¬«      }|j                  t        j                   j"                  j$                  k(  s=|j                  t        j                   j"                  j&                  k(  r<| j(                  s0|j[                  dd|	id|i|	j                  |j                  dœ¬«       ||_.        |S )aX  Initialize the input tensor with Xavier initialization.

        Args:
            var(Tensor): Tensor that needs to be initialized.
            block(Block, optional): The block in which initialization ops
                   should be added. Used in static graph only, default None.

        Returns:
            The initialization op
        r   NÚOut)Úuint16Úfloat16Úfloat32Úfloat64Úxavier_initÚ.ÚtmpF)ÚnameÚshapeÚdtypeÚtypeÚpersistableg      @g       @g        Úuniform_random)r*   r+   ÚminÚmaxr   T)r,   ÚinputsÚoutputsÚattrsÚstop_gradientÚgaussian_random)r*   r+   ÚmeanÚstdr   )r,   r2   r3   r4   ÚcastÚX)Úin_dtypeÚ	out_dtype)r,   r1   r2   r3   )/ÚpaddleÚ_check_blockÚ
isinstancer   ÚBlockÚpirr   ÚParameterMetar   Ú_compute_fansr   r   r   ÚprogramÚrandom_seedr+   ÚVarDescÚVarTypeÚFP16ÚBF16r   ÚFP32Ú
create_varr   ÚgenerateÚjoinr)   r*   Ú
LOD_TENSORÚDataTypeÚFLOAT16ÚBFLOAT16ÚFLOAT32r
   ÚmathÚsqrtÚfloatr   r   r	   Úgaussianr8   Ú_share_underline_tensor_tor   Ú_pir_opsÚ	append_opÚop)r   ÚvarÚblockr<   Úf_inÚf_outr   r   r;   Úout_varÚlimitr7   ÚplaceÚvar_tmprY   s                  r   ÚforwardzXavierInitializer.forwardL   s  € ó 	à×!Ñ! %Ó(ˆÜ˜%¤)§/¡/°6·:±:×3CÑ3CÐ!DÔEÐEÐEÜ˜#˜vŸz™zŸ™×<Ñ<Ô=Ü$ØØÚ;Øô	ð ×(Ñ(¨Ó-‰ˆˆeð Ÿ™Ð-‘°4·<±<ˆØŸ=™=Ð0‘%°d·m±mˆà�:‰:˜Š?ØŸ™×2Ñ2ˆDŒJð �9‰9œŸ™×,Ñ,×1Ñ1Ò1Ø�I‰IœŸ™×-Ñ-×2Ñ2Ò2¸4¿=º=äŸ™×,Ñ,×1Ñ1ˆIØ×&Ñ&Ü ×)Ñ)Ø—H‘H˜m¨S¯X©X°uÐ=Ó>óð —i‘iØÜ—\‘\×)Ñ)×4Ñ4Ø!ð 'ó ‰Gð �I‰Iœ$Ÿ-™-×/Ñ/´·±×1GÑ1GÐHÑHØ—M’MäŸ™×-Ñ-ˆIØ‰GàŸ	™	ˆIØˆGäÕØ�}Š}ÜŸ	™	 #¬¨f°wÑ.>Ó(?Ñ"?Ó@�Ü Ÿ.™.Ø—M‘MØØ�FØØ—J‘JÜ+Ó-ó‘ô —i‘i ¤e¨F°WÑ,<Ó&=Ñ =Ó>�ä/Ó1�Ü Ÿ/™/Ø—M‘M 3¨¨T¯Z©Z¸ÀEó�ð �y‰yœDŸL™L×0Ñ0×5Ñ5Ò5Ø—	‘	œTŸ\™\×1Ñ1×6Ñ6Ò6¸t¿}º}ä Ÿ+™+ g¨s¯y©yÓ9�Ø×2Ñ2°3Ô7ð ð ×2Ñ2°3Ô7ØÜ�]Ø�}Š}ÜŸ	™	 #¬¨f°wÑ.>Ó(?Ñ"?Ó@�Ø Ÿ/™/×1Ñ1Ø—M‘MØØ�FØØ—J‘JÜ+Ó-ó‘ô —i‘i ¤e¨F°WÑ,<Ó&=Ñ =Ó>�Ü Ÿ/™/Ø—M‘MØØØ—J‘JØÜ+Ó-ó�ð —	‘	œdŸm™m×3Ñ3´T·]±]×5KÑ5KÐLÑLØŸšä—{‘{ 7¨C¯I©IÓ6Ð6àˆNà�}Š}ÜŸ	™	 #¬¨f°wÑ.>Ó(?Ñ"?Ó@�Ø—_‘_Ø)ØØ" GÐ,à!(§¡Ø!*Ø %˜vØ$Ø $§
¡
ñð #'ð %ó ‘ô —i‘i ¤e¨F°WÑ,<Ó&=Ñ =Ó>�Ø—_‘_Ø*Ø" GÐ,à!(§¡Ø!(§¡Ø #Ø"Ø $§
¡
ñð #'ð %ó �ð �y‰yœDŸL™L×0Ñ0×5Ñ5Ò5Ø—	‘	œTŸ\™\×1Ñ1×6Ñ6Ò6¸t¿}º}à—‘ØØ ˜>Ø" C˜LØ'.§}¡}À3Ç9Á9ÑMð	  ô ð ˆCŒFØˆIr   )TNNr   r   )Ú__name__Ú
__module__Ú__qualname__Ú__doc__r   rb   Ú__classcell__©r   s   @r   r   r      s   ø„ ñ!õF÷Zr   r   c                   ó$   ‡ — e Zd ZdZdˆ fd„	Zˆ xZS )ÚXavierNormalaµ  
    This class implements the Xavier weight initializer from the paper
    `Understanding the difficulty of training deep feedforward neural
    networks <http://proceedings.mlr.press/v9/glorot10a/glorot10a.pdf>`_
    by Xavier Glorot and Yoshua Bengio, using a normal distribution whose mean is :math:`0` and standard deviation is

    .. math::

        \sqrt{\frac{2.0}{fan\_in + fan\_out}}.


    Args:
        fan_in (float, optional): fan_in for Xavier initialization, which is
                inferred from the Tensor. Default is None.
        fan_out (float, optional): fan_out for Xavier initialization, which is
                 inferred from the Tensor. Default is None.
        name (str, optional): For details, please refer to :ref:`api_guide_Name`. Generally, no setting is required. Default: None.

    Returns:
        A parameter initialized by Xavier weight, using a normal distribution.

    Examples:
        .. code-block:: python

            >>> import paddle
            >>> paddle.seed(1)
            >>> data = paddle.ones(shape=[3, 1, 2], dtype='float32')
            >>> weight_attr = paddle.framework.ParamAttr(
            ...     name="linear_weight",
            ...     initializer=paddle.nn.initializer.XavierNormal())
            >>> bias_attr = paddle.framework.ParamAttr(
            ...     name="linear_bias",
            ...     initializer=paddle.nn.initializer.XavierNormal())
            >>> linear = paddle.nn.Linear(2, 2, weight_attr=weight_attr, bias_attr=bias_attr)
            >>> print(linear.weight)
            Parameter containing:
            Tensor(shape=[2, 2], dtype=float32, place=Place(cpu), stop_gradient=False,
            [[-0.21607460,  0.08382989],
             [ 0.29147008, -0.07049121]])

            >>> print(linear.bias)
            Parameter containing:
            Tensor(shape=[2], dtype=float32, place=Place(cpu), stop_gradient=False,
            [1.06076419, 0.87684733])

            >>> res = linear(data)
            >>> print(res)
            Tensor(shape=[3, 1, 2], dtype=float32, place=Place(cpu), stop_gradient=False,
            [[[1.13615966, 0.89018601]],
             [[1.13615966, 0.89018601]],
             [[1.13615966, 0.89018601]]])
    c                 ó,   •— t         ‰| �  d||d¬«       y )NFr   ©r   r   r   r   ©r   r   ©r   r   r   r)   r   s       €r   r   zXavierNormal.__init__  s   ø€ Ü‰Ñ ¨v¸wÈQÐÕOr   ©NNN©rc   rd   re   rf   r   rg   rh   s   @r   rj   rj   é   s   ø„ ñ3÷jPñ Pr   rj   c                   ó$   ‡ — e Zd ZdZdˆ fd„	Zˆ xZS )ÚXavierUniforma+	  
    This class implements the Xavier weight initializer from the paper
    `Understanding the difficulty of training deep feedforward neural
    networks <http://proceedings.mlr.press/v9/glorot10a/glorot10a.pdf>`_
    by Xavier Glorot and Yoshua Bengio.

    This initializer is designed to keep the scale of the gradients
    approximately same in all the layers. In case of Uniform distribution,
    the range is :math:`[-x,x]`, where

    .. math::

        x = \sqrt{\frac{6.0}{fan\_in + fan\_out}}.

    Args:
        fan_in (float, optional): fan_in for Xavier initialization, which is
                inferred from the Tensor. Default is None.
        fan_out (float, optional): fan_out for Xavier initialization, which is
                 inferred from the Tensor. Default is None.
        name (str, optional): For details, please refer to :ref:`api_guide_Name`. Generally, no setting is required. Default: None.

    Returns:
        A parameter initialized by Xavier weight, using a uniform distribution.

    Examples:
        .. code-block:: python

            >>> import paddle
            >>> paddle.seed(1)
            >>> data = paddle.ones(shape=[3, 1, 2], dtype='float32')
            >>> weight_attr = paddle.framework.ParamAttr(
            ...     name="linear_weight",
            ...     initializer=paddle.nn.initializer.XavierUniform())
            >>> bias_attr = paddle.framework.ParamAttr(
            ...     name="linear_bias",
            ...     initializer=paddle.nn.initializer.XavierUniform())
            >>> linear = paddle.nn.Linear(2, 2, weight_attr=weight_attr, bias_attr=bias_attr)
            >>> print(linear.weight)
            Parameter containing:
            Tensor(shape=[2, 2], dtype=float32, place=Place(cpu), stop_gradient=False,
            [[-1.18095720,  0.64892638],
             [ 0.43125069, -1.11156428]])
            >>> print(linear.bias)
            Parameter containing:
            Tensor(shape=[2], dtype=float32, place=Place(cpu), stop_gradient=False,
            [-0.27524316,  1.13808715])

            >>> res = linear(data)
            >>> print(res)
            Tensor(shape=[3, 1, 2], dtype=float32, place=Place(cpu), stop_gradient=False,
            [[[-1.02494967,  0.67544925]],
             [[-1.02494967,  0.67544925]],
             [[-1.02494967,  0.67544925]]])
    c                 ó,   •— t         ‰| �  d||d¬«       y )NTr   rl   rm   rn   s       €r   r   zXavierUniform.__init__[  s   ø€ Ü‰Ñ ¨f¸gÈAÐÕNr   ro   rp   rh   s   @r   rr   rr   #  s   ø„ ñ5÷nOñ Or   rr   )rR   r<   r   Úbaser   r   r   Úbase.data_feederr   Úbase.frameworkr	   r
   r   Úinitializerr   Ú__all__r   rj   rr   © r   r   Ú<module>rz      sX   ðó å ç 0Ñ 0Ý 8÷ñ õ
 %à
€ôG˜ô GôT7PÐ$ô 7Pôt9OÐ%õ 9Or   