§
    �Štj4>  ã                   ó
  — d dl Z d dlZd dlmZmZ d dlmZmZ d dlm	Z	 ddl
mZ ddlmZ g d¢Z G d	„ d
e¦  «        Z G d„ de¦  «        Zeee         z  ez  Z G d„ de¦  «        Z G d„ de¦  «        Z G d„ de¦  «        ZdS )é    N)ÚSizeÚTensor)Ú
functionalÚinit)Ú	Parameteré   )ÚCrossMapLRN2d)ÚModule)ÚLocalResponseNormr	   Ú	LayerNormÚ	GroupNormÚRMSNormc                   ó„   ‡ — e Zd ZU dZg d¢Zeed<   eed<   eed<   eed<   	 ddedededed
df
ˆ fd„Zde	d
e	fd„Z
d„ Zˆ xZS )r   a‹  Applies local response normalization over an input signal.

    The input signal is composed of several input planes, where channels occupy the second dimension.
    Applies normalization across channels.

    .. math::
        b_{c} = a_{c}\left(k + \frac{\alpha}{n}
        \sum_{c'=\max(0, c-n/2)}^{\min(N-1,c+n/2)}a_{c'}^2\right)^{-\beta}

    Args:
        size: amount of neighbouring channels used for normalization
        alpha: multiplicative factor. Default: 0.0001
        beta: exponent. Default: 0.75
        k: additive factor. Default: 1

    Shape:
        - Input: :math:`(N, C, *)`
        - Output: :math:`(N, C, *)` (same shape as input)

    Examples::

        >>> lrn = nn.LocalResponseNorm(2)
        >>> signal_2d = torch.randn(32, 5, 24, 24)
        >>> signal_4d = torch.randn(16, 5, 7, 7, 7, 7)
        >>> output_2d = lrn(signal_2d)
        >>> output_4d = lrn(signal_4d)

    )ÚsizeÚalphaÚbetaÚkr   r   r   r   ç-Cëâ6?ç      è?ç      ð?ÚreturnNc                 ó€   •— t          ¦   «                              ¦   «          || _        || _        || _        || _        d S ©N©ÚsuperÚ__init__r   r   r   r   ©Úselfr   r   r   r   Ú	__class__s        €ú\/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/torch/nn/modules/normalization.pyr   zLocalResponseNorm.__init__4   ó;   ø€ õ 	‰Œ×ÒÑÔÐØˆŒ	ØˆŒ
ØˆŒ	ØˆŒˆˆó    Úinputc                 óZ   — t          j        || j        | j        | j        | j        ¦  «        S ©z(
        Runs the forward pass.
        )ÚFÚlocal_response_normr   r   r   r   ©r   r#   s     r    ÚforwardzLocalResponseNorm.forward=   s%   € õ Ô$ U¨D¬I°t´zÀ4Ä9ÈdÌfÑUÔUÐUr"   c                 ó&   —  dj         di | j        ¤ŽS ©ú@
        Return the extra representation of the module.
        z){size}, alpha={alpha}, beta={beta}, k={k}© ©ÚformatÚ__dict__©r   s    r    Ú
extra_reprzLocalResponseNorm.extra_reprC   ó!   € ð BÐ:ÔAÐRÐRÀDÄMÐRÐRÐRr"   )r   r   r   )Ú__name__Ú
__module__Ú__qualname__Ú__doc__Ú__constants__ÚintÚ__annotations__Úfloatr   r   r)   r2   Ú__classcell__©r   s   @r    r   r      sì   ø€ € € € € € ðð ð: 3Ð2Ð2€MØ
€I€I�IØ€L€L�LØ
€K€K�KØ€H€H�Hð NQðð ØðØ %ðØ49ðØEJðà	ðð ð ð ð ð ðV˜Vð V¨ð Vð Vð Vð VðSð Sð Sð Sð Sð Sð Sr"   r   c                   ó~   ‡ — e Zd ZU eed<   eed<   eed<   eed<   	 ddededededd	f
ˆ fd
„Zdedefd„Zde	fd„Z
ˆ xZS )r	   r   r   r   r   r   r   r   r   Nc                 ó€   •— t          ¦   «                              ¦   «          || _        || _        || _        || _        d S r   r   r   s        €r    r   zCrossMapLRN2d.__init__P   r!   r"   r#   c                 óZ   — t          j        || j        | j        | j        | j        ¦  «        S r%   )Ú_cross_map_lrn2dÚapplyr   r   r   r   r(   s     r    r)   zCrossMapLRN2d.forwardY   s%   € õ  Ô% e¨T¬Y¸¼
ÀDÄIÈtÌvÑVÔVÐVr"   c                 ó&   —  dj         di | j        ¤ŽS r+   r.   r1   s    r    r2   zCrossMapLRN2d.extra_repr_   r3   r"   )r   r   r   )r4   r5   r6   r9   r:   r;   r   r   r)   Ústrr2   r<   r=   s   @r    r	   r	   J   sã   ø€ € € € € € Ø
€I€I�IØ€L€L�LØ
€K€K�KØ€H€H�Hð NOðð ØðØ %ðØ49ðØEJðà	ðð ð ð ð ð ðW˜Vð W¨ð Wð Wð Wð WðS˜Cð Sð Sð Sð Sð Sð Sð Sð Sr"   r	   c                   ó    ‡ — e Zd ZU dZg d¢Zeedf         ed<   eed<   e	ed<   	 	 	 	 	 dde
dede	d
e	dd	f
ˆ fd„Zdd„Zdedefd„Zdefd„Zˆ xZS )r   a¸  Applies Layer Normalization over a mini-batch of inputs.

    This layer implements the operation as described in
    the paper `Layer Normalization <https://arxiv.org/abs/1607.06450>`__

    .. math::
        y = \frac{x - \mathrm{E}[x]}{ \sqrt{\mathrm{Var}[x] + \epsilon}} * \gamma + \beta

    The mean and standard-deviation are calculated over the last `D` dimensions, where `D`
    is the dimension of :attr:`normalized_shape`. For example, if :attr:`normalized_shape`
    is ``(3, 5)`` (a 2-dimensional shape), the mean and standard-deviation are computed over
    the last 2 dimensions of the input (i.e. ``input.mean((-2, -1))``).
    :math:`\gamma` and :math:`\beta` are learnable affine transform parameters of
    :attr:`normalized_shape` if :attr:`elementwise_affine` is ``True``.
    The variance is calculated via the biased estimator, equivalent to
    `torch.var(input, correction=0)`.

    .. note::
        Unlike Batch Normalization and Instance Normalization, which applies
        scalar scale and bias for each entire channel/plane with the
        :attr:`affine` option, Layer Normalization applies per-element scale and
        bias with :attr:`elementwise_affine`.

    This layer uses statistics computed from input data in both training and
    evaluation modes.

    Args:
        normalized_shape (int or list or torch.Size): input shape from an expected input
            of size

            .. math::
                [* \times \text{normalized\_shape}[0] \times \text{normalized\_shape}[1]
                    \times \ldots \times \text{normalized\_shape}[-1]]

            If a single integer is used, it is treated as a singleton list, and this module will
            normalize over the last dimension which is expected to be of that specific size.
        eps: a value added to the denominator for numerical stability. Default: 1e-5
        elementwise_affine: a boolean value that when set to ``True``, this module
            has learnable per-element affine parameters initialized to ones (for weights)
            and zeros (for biases). Default: ``True``
        bias: If set to ``False``, the layer will not learn an additive bias (only relevant if
            :attr:`elementwise_affine` is ``True``). Default: ``True``

    Attributes:
        weight: the learnable weights of the module of shape
            :math:`\text{normalized\_shape}` when :attr:`elementwise_affine` is set to ``True``.
            The values are initialized to 1.
        bias:   the learnable bias of the module of shape
                :math:`\text{normalized\_shape}` when :attr:`elementwise_affine` is set to ``True``.
                The values are initialized to 0.

    Shape:
        - Input: :math:`(N, *)`
        - Output: :math:`(N, *)` (same shape as input)

    Examples::

        >>> # NLP Example
        >>> batch, sentence_length, embedding_dim = 20, 5, 10
        >>> embedding = torch.randn(batch, sentence_length, embedding_dim)
        >>> layer_norm = nn.LayerNorm(embedding_dim)
        >>> # Activate module
        >>> layer_norm(embedding)
        >>>
        >>> # Image Example
        >>> N, C, H, W = 20, 5, 10, 10
        >>> input = torch.randn(N, C, H, W)
        >>> # Normalize over the last three dimensions (i.e. the channel and spatial dimensions)
        >>> # as shown in the image below
        >>> layer_norm = nn.LayerNorm([C, H, W])
        >>> output = layer_norm(input)

    .. image:: ../_static/img/nn/layer_norm.jpg
        :scale: 50 %

    ©Únormalized_shapeÚepsÚelementwise_affine.rG   rH   rI   çñhãˆµøä>TNÚbiasr   c                 ó6  •— ||dœ}t          ¦   «                              ¦   «          t          |t          j        ¦  «        r|f}t          |¦  «        | _        || _        || _        | j        rlt          t          j        | j        fi |¤Ž¦  «        | _        |r*t          t          j        | j        fi |¤Ž¦  «        | _        nC|                      dd ¦  «         n,|                      dd ¦  «         |                      dd ¦  «         |                      ¦   «          d S )N©ÚdeviceÚdtyperK   Úweight)r   r   Ú
isinstanceÚnumbersÚIntegralÚtuplerG   rH   rI   r   ÚtorchÚemptyrP   rK   Úregister_parameterÚreset_parameters)	r   rG   rH   rI   rK   rN   rO   Úfactory_kwargsr   s	           €r    r   zLayerNorm.__init__¼   s/  ø€ ð %+°UÐ;Ð;ˆÝ‰Œ×ÒÑÔÐÝÐ&­Ô(8Ñ9Ô9ð 	3à 0Ð2ÐÝ %Ð&6Ñ 7Ô 7ˆÔØˆŒØ"4ˆÔØÔ"ð 	2Ý#Ý”˜DÔ1ÐDÐD°^ÐDÐDñô ˆDŒKð ð 6Ý%Ý”K Ô 5ÐHÐH¸ÐHÐHñô �”	�	ð ×'Ò'¨°Ñ5Ô5Ð5Ð5à×#Ò# H¨dÑ3Ô3Ð3Ø×#Ò# F¨DÑ1Ô1Ð1à×ÒÑÔÐÐÐr"   c                 óŽ   — | j         r;t          j        | j        ¦  «         | j        �t          j        | j        ¦  «         d S d S d S r   )rI   r   Úones_rP   rK   Úzeros_r1   s    r    rX   zLayerNorm.reset_parametersÝ   sO   € ØÔ"ð 	'ÝŒJ�t”{Ñ#Ô#Ð#ØŒyÐ$Ý”˜DœIÑ&Ô&Ð&Ð&Ð&ð	'ð 	'à$Ð$r"   r#   c                 óZ   — t          j        || j        | j        | j        | j        ¦  «        S r   )r&   Ú
layer_normrG   rP   rK   rH   r(   s     r    r)   zLayerNorm.forwardã   s*   € ÝŒ|Ø�4Ô(¨$¬+°t´yÀ$Ä(ñ
ô 
ð 	
r"   c                 ó<   —  dj         di | j        ¤d| j        d ui¤ŽS )NzW{normalized_shape}, eps={eps}, elementwise_affine={elementwise_affine}, bias={use_bias}Úuse_biasr-   ©r/   r0   rK   r1   s    r    r2   zLayerNorm.extra_reprè   óN   € ð%ð Ü$ðVð VØ'+¤}ðVð VØ?C¼yÐPTÐ?TðVð Vð Vð	
r"   )rJ   TTNN©r   N)r4   r5   r6   r7   r8   rT   r9   r:   r;   ÚboolÚ_shape_tr   rX   r   r)   rD   r2   r<   r=   s   @r    r   r   i   s  ø€ € € € € € ðKð KðZ FÐEÐE€MØ˜C ˜H”oÐ%Ð%Ñ%Ø	€J€J�JØÐÐÑð
 Ø#'ØØØð ð  à"ð ð ð ð !ð	 ð
 ð ð 
ð ð  ð  ð  ð  ð  ðB'ð 'ð 'ð 'ð
˜Vð 
¨ð 
ð 
ð 
ð 
ð

˜Cð 
ð 
ð 
ð 
ð 
ð 
ð 
ð 
r"   r   c                   ó¢   ‡ — e Zd ZU dZg d¢Zeed<   eed<   eed<   eed<   	 	 	 	 ddd
œdedededededd	fˆ fd„Z	dd„Z
dedefd„Zdefd„Zˆ xZS )r   aR  Applies Group Normalization over a mini-batch of inputs.

    This layer implements the operation as described in
    the paper `Group Normalization <https://arxiv.org/abs/1803.08494>`__

    .. math::
        y = \frac{x - \mathrm{E}[x]}{ \sqrt{\mathrm{Var}[x] + \epsilon}} * \gamma + \beta

    The input channels are separated into :attr:`num_groups` groups, each containing
    ``num_channels / num_groups`` channels. :attr:`num_channels` must be divisible by
    :attr:`num_groups`. The mean and standard-deviation are calculated
    separately over each group. :math:`\gamma` and :math:`\beta` are learnable
    per-channel affine transform parameter vectors of size :attr:`num_channels` if
    :attr:`affine` is ``True``.
    The variance is calculated via the biased estimator, equivalent to
    `torch.var(input, correction=0)`.

    This layer uses statistics computed from input data in both training and
    evaluation modes.

    Args:
        num_groups (int): number of groups to separate the channels into
        num_channels (int): number of channels expected in input
        eps: a value added to the denominator for numerical stability. Default: 1e-5
        affine: a boolean value that when set to ``True``, this module
            has learnable per-channel affine parameters initialized to ones (for weights)
            and zeros (for biases). Default: ``True``
        bias: If set to ``False``, the layer will not learn an additive bias (only relevant if
            :attr:`affine` is ``True``). Default: ``True``

    Shape:
        - Input: :math:`(N, C, *)` where :math:`C=\text{num\_channels}`
        - Output: :math:`(N, C, *)` (same shape as input)

    Examples::

        >>> input = torch.randn(20, 6, 10, 10)
        >>> # Separate 6 channels into 3 groups
        >>> m = nn.GroupNorm(3, 6)
        >>> # Separate 6 channels into 6 groups (equivalent with InstanceNorm)
        >>> m = nn.GroupNorm(6, 6)
        >>> # Put all 6 channels into a single group (equivalent with LayerNorm)
        >>> m = nn.GroupNorm(1, 6)
        >>> # Activating the module
        >>> output = m(input)
    )Ú
num_groupsÚnum_channelsrH   Úaffinerg   rh   rH   ri   rJ   TN)rK   rK   r   c                ó  •— ||dœ}t          ¦   «                              ¦   «          ||z  dk    rt          d|› d|› d�¦  «        ‚|| _        || _        || _        || _        | j        rbt          t          j	        |fi |¤Ž¦  «        | _
        |r%t          t          j	        |fi |¤Ž¦  «        | _        nC|                      dd ¦  «         n,|                      dd ¦  «         |                      dd ¦  «         |                      ¦   «          d S )NrM   r   znum_channels (z#) must be divisible by num_groups (ú)rK   rP   )r   r   Ú
ValueErrorrg   rh   rH   ri   r   rU   rV   rP   rK   rW   rX   )
r   rg   rh   rH   ri   rN   rO   rK   rY   r   s
            €r    r   zGroupNorm.__init__%  s3  ø€ ð %+°UÐ;Ð;ˆÝ‰Œ×ÒÑÔÐØ˜*Ñ$¨Ò)Ð)ÝØ_ Ð_Ð_ÐR\Ð_Ð_Ð_ñô ð ð %ˆŒØ(ˆÔØˆŒØˆŒØŒ;ð 	2Ý#¥E¤K°Ð$OÐ$OÀÐ$OÐ$OÑPÔPˆDŒKØð 6Ý%¥e¤k°,Ð&QÐ&QÀ.Ð&QÐ&QÑRÔR�”	�	à×'Ò'¨°Ñ5Ô5Ð5Ð5à×#Ò# H¨dÑ3Ô3Ð3Ø×#Ò# F¨DÑ1Ô1Ð1à×ÒÑÔÐÐÐr"   c                 óŽ   — | j         r;t          j        | j        ¦  «         | j        �t          j        | j        ¦  «         d S d S d S r   )ri   r   r[   rP   rK   r\   r1   s    r    rX   zGroupNorm.reset_parametersG  sN   € ØŒ;ð 	'ÝŒJ�t”{Ñ#Ô#Ð#ØŒyÐ$Ý”˜DœIÑ&Ô&Ð&Ð&Ð&ð	'ð 	'à$Ð$r"   r#   c                 óZ   — t          j        || j        | j        | j        | j        ¦  «        S r   )r&   Ú
group_normrg   rP   rK   rH   r(   s     r    r)   zGroupNorm.forwardM  s"   € ÝŒ|˜E 4¤?°D´KÀÄÈDÌHÑUÔUÐUr"   c                 ó<   —  dj         di | j        ¤d| j        d ui¤ŽS )NzI{num_groups}, {num_channels}, eps={eps}, affine={affine}, bias={use_bias}r`   r-   ra   r1   s    r    r2   zGroupNorm.extra_reprP  rb   r"   )rJ   TNNrc   )r4   r5   r6   r7   r8   r9   r:   r;   rd   r   rX   r   r)   rD   r2   r<   r=   s   @r    r   r   ï   s3  ø€ € € € € € ð-ð -ð^ DÐCÐC€MØ€O€O�OØÐÐÑØ	€J€J�JØ€L€L�Lð ØØØð  ð ð  ð   ð   àð  ð ð  ð ð	  ð
 ð  ð ð  ð 
ð  ð   ð   ð   ð   ð   ðD'ð 'ð 'ð 'ðV˜Vð V¨ð Vð Vð Vð Vð
˜Cð 
ð 
ð 
ð 
ð 
ð 
ð 
ð 
r"   r   c            	       óº   ‡ — e Zd ZU dZg d¢Zeedf         ed<   edz  ed<   e	ed<   	 	 	 	 dde
dedz  de	d	dfˆ fd
„Zdd„Zdej        d	ej        fd„Zd	efd„Zˆ xZS )r   a‚  Applies Root Mean Square Layer Normalization over a mini-batch of inputs.

    This layer implements the operation as described in
    the paper `Root Mean Square Layer Normalization <https://arxiv.org/pdf/1910.07467.pdf>`__

    .. math::
        y_i = \frac{x_i}{\mathrm{RMS}(x)} * \gamma_i, \quad
        \text{where} \quad \text{RMS}(x) = \sqrt{\epsilon + \frac{1}{n} \sum_{i=1}^{n} x_i^2}

    The RMS is taken over the last ``D`` dimensions, where ``D``
    is the dimension of :attr:`normalized_shape`. For example, if :attr:`normalized_shape`
    is ``(3, 5)`` (a 2-dimensional shape), the RMS is computed over
    the last 2 dimensions of the input.

    Args:
        normalized_shape (int or list or torch.Size): input shape from an expected input
            of size

            .. math::
                [* \times \text{normalized\_shape}[0] \times \text{normalized\_shape}[1]
                    \times \ldots \times \text{normalized\_shape}[-1]]

            If a single integer is used, it is treated as a singleton list, and this module will
            normalize over the last dimension which is expected to be of that specific size.
        eps (float, optional): a value added to the denominator for numerical stability.
            If not specified, uses the machine epsilon of the computation (opmath) type:
            fp16/bf16 and fp32 inputs use ``torch.finfo(torch.float32).eps``, while fp64
            inputs use ``torch.finfo(torch.float64).eps``. Default: ``None``
        elementwise_affine: a boolean value that when set to ``True``, this module
            has learnable per-element affine parameters initialized to ones (for weights). Default: ``True``.

    Shape:
        - Input: :math:`(N, *)`
        - Output: :math:`(N, *)` (same shape as input)

    Examples::

        >>> rms_norm = nn.RMSNorm([2, 3])
        >>> input = torch.randn(2, 2, 3)
        >>> rms_norm(input)

    rF   .rG   NrH   rI   Tr   c                 ó†  •— ||dœ}t          ¦   «                              ¦   «          t          |t          j        ¦  «        r|f}t          |¦  «        | _        || _        || _        | j        r*t          t          j        | j        fi |¤Ž¦  «        | _        n|                      dd ¦  «         |                      ¦   «          d S )NrM   rP   )r   r   rQ   rR   rS   rT   rG   rH   rI   r   rU   rV   rP   rW   rX   )r   rG   rH   rI   rN   rO   rY   r   s          €r    r   zRMSNorm.__init__ˆ  sÍ   ø€ ð %+°UÐ;Ð;ˆÝ‰Œ×ÒÑÔÐÝÐ&­Ô(8Ñ9Ô9ð 	3à 0Ð2ÐÝ %Ð&6Ñ 7Ô 7ˆÔØˆŒØ"4ˆÔØÔ"ð 	4Ý#Ý”˜DÔ1ÐDÐD°^ÐDÐDñô ˆDŒKˆKð ×#Ò# H¨dÑ3Ô3Ð3Ø×ÒÑÔÐÐÐr"   c                 óJ   — | j         rt          j        | j        ¦  «         dS dS )zS
        Resets parameters based on their initialization used in __init__.
        N)rI   r   r[   rP   r1   s    r    rX   zRMSNorm.reset_parameters   s1   € ð Ô"ð 	$ÝŒJ�t”{Ñ#Ô#Ð#Ð#Ð#ð	$ð 	$r"   Úxc                 óN   — t          j        || j        | j        | j        ¦  «        S r%   )r&   Úrms_normrG   rP   rH   )r   rt   s     r    r)   zRMSNorm.forward§  s!   € õ Œz˜!˜TÔ2°D´KÀÄÑJÔJÐJr"   c                 ó&   —  dj         di | j        ¤ŽS )r,   zF{normalized_shape}, eps={eps}, elementwise_affine={elementwise_affine}r-   r.   r1   s    r    r2   zRMSNorm.extra_repr­  s1   € ð
=ð 6Ü6<ðNð NØ?C¼}ðNð Nð	
r"   )NTNNrc   )r4   r5   r6   r7   r8   rT   r9   r:   r;   rd   re   r   rX   rU   r   r)   rD   r2   r<   r=   s   @r    r   r   W  s   ø€ € € € € € ð)ð )ðV FÐEÐE€MØ˜C ˜H”oÐ%Ð%Ñ%Ø	�‰ÐÐÑØÐÐÑð
 !Ø#'ØØð ð  à"ð ð �T‰\ð ð !ð	 ð 
ð ð  ð  ð  ð  ð  ð0$ð $ð $ð $ðK˜œð K¨%¬,ð Kð Kð Kð Kð
˜Cð 
ð 
ð 
ð 
ð 
ð 
ð 
ð 
r"   r   )rR   rU   r   r   Útorch.nnr   r&   r   Útorch.nn.parameterr   Ú
_functionsr	   rA   Úmoduler
   Ú__all__r   r9   Úlistre   r   r   r   r-   r"   r    ú<module>r~      s¨  ðà €€€à €€€Ø Ð Ð Ð Ð Ð Ð Ð Ø *Ð *Ð *Ð *Ð *Ð *Ð *Ð *Ø (Ð (Ð (Ð (Ð (Ð (à 9Ð 9Ð 9Ð 9Ð 9Ð 9Ø Ð Ð Ð Ð Ð ð VÐ
UÐ
U€ð7Sð 7Sð 7Sð 7Sð 7S˜ñ 7Sô 7Sð 7SðtSð Sð Sð Sð S�Fñ Sô Sð Sð8 ��c”‰?˜TÑ!€ðC
ð C
ð C
ð C
ð C
�ñ C
ô C
ð C
ðLe
ð e
ð e
ð e
ð e
�ñ e
ô e
ð e
ðP]
ð ]
ð ]
ð ]
ð ]
ˆfñ ]
ô ]
ð ]
ð ]
ð ]
r"   