§
    fŠtjz>  ã                   óú  — d Z ddlZddlZddlmZ ddlmZ ddlm	Z	m
Z
mZmZmZmZ ddgZ e¦   «          ed	„ d
„ dd„ dd¬¦  «        	 	 	 d$dej        j        dej        j        dz  dedz  dedej        ej        z  f
d„¦   «         ¦   «         Zd%d„Z e¦   «          ed„ dd„ e¬¦  «        dddddœdej        j        dedz  dedz  dededej        ej        z  fd„¦   «         ¦   «         Zd„ Zd „ Zd!„ Zd"„ Zd#„ ZdS )&z5
Created on Fri Apr  2 09:06:05 2021

@author: matth
é    N)Úspecialé   )Ú_axis_nan_policy_factory)Úarray_namespaceÚ
xp_promoteÚ	xp_deviceÚ	is_marrayÚ_share_masksÚxp_capabilitiesÚentropyÚdifferential_entropyc                 ó   — | S ©N© ©Úxs    úR/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/scipy/stats/_entropy.pyú<lambda>r      ó   € ˆa€ ó    c                 ó"   — d| v r
| d         �dndS )NÚqké   r   r   )Úkwgss    r   r   r      s    € Ø�dˆlˆl˜t DœzÐ5ˆˆØð r   c                 ó   — | fS r   r   ©r   Ú_s     r   r   r      s   € ¨q¨d€ r   Téÿÿÿÿ)Ú	n_samplesÚ	n_outputsÚresult_to_tupleÚpairedÚ	too_smallÚpkr   ÚbaseÚaxisÚreturnc                 óÂ  — |�|dk    rt          d¦  «        ‚t          | |¦  «        }t          | |d|¬¦  «        \  } }t          j        d¬¦  «        5  |�0t          | ||¬¦  «        \  } }||                     ||d¬	¦  «        z  }| |                     | |d¬	¦  «        z  } ddd¦  «         n# 1 swxY w Y   |€t          j        | ¦  «        }n`t          |¦  «        r<t          j
        | j        |j        ¦  «        }|                     || j        ¬
¦  «        }nt          j
        | |¦  «        }|                     ||¬¦  «        }|�|t          j        |¦  «        z  }|S )aô  
    Calculate the Shannon entropy/relative entropy of given distribution(s).

    If only probabilities `pk` are given, the Shannon entropy is calculated as
    ``H = -sum(pk * log(pk))``.

    If `qk` is not None, then compute the relative entropy
    ``D = sum(pk * log(pk / qk))``. This quantity is also known
    as the Kullback-Leibler divergence.

    This routine will normalize `pk` and `qk` if they don't sum to 1.

    Parameters
    ----------
    pk : array_like
        Defines the (discrete) distribution. Along each axis-slice of ``pk``,
        element ``i`` is the  (possibly unnormalized) probability of event
        ``i``.
    qk : array_like, optional
        Sequence against which the relative entropy is computed. Should be in
        the same format as `pk`.
    base : float, optional
        The logarithmic base to use, defaults to ``e`` (natural logarithm).
    axis : int, optional
        The axis along which the entropy is calculated. Default is 0.

    Returns
    -------
    S : {float, array_like}
        The calculated entropy.

    Notes
    -----
    Informally, the Shannon entropy quantifies the expected uncertainty
    inherent in the possible outcomes of a discrete random variable.
    For example,
    if messages consisting of sequences of symbols from a set are to be
    encoded and transmitted over a noiseless channel, then the Shannon entropy
    ``H(pk)`` gives a tight lower bound for the average number of units of
    information needed per symbol if the symbols occur with frequencies
    governed by the discrete distribution `pk` [1]_. The choice of base
    determines the choice of units; e.g., ``e`` for nats, ``2`` for bits, etc.

    The relative entropy, ``D(pk|qk)``, quantifies the increase in the average
    number of units of information needed per symbol if the encoding is
    optimized for the probability distribution `qk` instead of the true
    distribution `pk`. Informally, the relative entropy quantifies the expected
    excess in surprise experienced if one believes the true distribution is
    `qk` when it is actually `pk`.

    A related quantity, the cross entropy ``CE(pk, qk)``, satisfies the
    equation ``CE(pk, qk) = H(pk) + D(pk|qk)`` and can also be calculated with
    the formula ``CE = -sum(pk * log(qk))``. It gives the average
    number of units of information needed per symbol if an encoding is
    optimized for the probability distribution `qk` when the true distribution
    is `pk`. It is not computed directly by `entropy`, but it can be computed
    using two calls to the function (see Examples).

    See [2]_ for more information.

    References
    ----------
    .. [1] Shannon, C.E. (1948), A Mathematical Theory of Communication.
           Bell System Technical Journal, 27: 379-423.
           https://doi.org/10.1002/j.1538-7305.1948.tb01338.x
    .. [2] Thomas M. Cover and Joy A. Thomas. 2006. Elements of Information
           Theory (Wiley Series in Telecommunications and Signal Processing).
           Wiley-Interscience, USA.


    Examples
    --------
    The outcome of a fair coin is the most uncertain:

    >>> import numpy as np
    >>> from scipy.stats import entropy
    >>> base = 2  # work in units of bits
    >>> pk = np.array([1/2, 1/2])  # fair coin
    >>> H = entropy(pk, base=base)
    >>> H
    1.0
    >>> H == -np.sum(pk * np.log(pk)) / np.log(base)
    True

    The outcome of a biased coin is less uncertain:

    >>> qk = np.array([9/10, 1/10])  # biased coin
    >>> entropy(qk, base=base)
    0.46899559358928117

    The relative entropy between the fair coin and biased coin is calculated
    as:

    >>> D = entropy(pk, qk, base=base)
    >>> D
    0.7369655941662062
    >>> np.isclose(D, np.sum(pk * np.log(pk/qk)) / np.log(base), rtol=4e-16, atol=0)
    True

    The cross entropy can be calculated as the sum of the entropy and
    relative entropy`:

    >>> CE = entropy(pk, base=base) + entropy(pk, qk, base=base)
    >>> CE
    1.736965594166206
    >>> CE == -np.sum(pk * np.log(qk)) / np.log(base)
    True

    Nr   ú+`base` must be a positive number or `None`.T)Ú	broadcastÚxpÚignore)Úinvalid©r+   ©r&   Úkeepdims)Úmask©r&   )Ú
ValueErrorr   r   ÚnpÚerrstater
   Úsumr   Úentrr	   Úrel_entrÚdataÚasarrayr1   ÚmathÚlog)r$   r   r%   r&   r+   ÚvecÚSs          r   r   r      s¡  € ðx Ð˜D AšI˜IÝÐFÑGÔGÐGå	˜˜RÑ	 Ô	 €BÝ˜˜B¨$°2Ð6Ñ6Ô6�F€Bˆå	Œ˜XÐ	&Ñ	&Ô	&ð 7ð 7Øˆ>Ý! " b¨RÐ0Ñ0Ô0‰FˆB�Ø�b—f’f˜R d°T�fÑ:Ô:Ñ:ˆBØ�"—&’&˜ $°�&Ñ6Ô6Ñ6ˆð	7ð 7ð 7ñ 7ô 7ð 7ð 7ð 7ð 7ð 7ð 7øøøð 7ð 7ð 7ð 7ð 
€zÝŒl˜2ÑÔˆˆå�R‰=Œ=ð 	+ÝÔ" 2¤7¨B¬GÑ4Ô4ˆCØ—*’*˜S r¤w�*Ñ/Ô/ˆCˆCåÔ" 2 rÑ*Ô*ˆCà
�Šˆs˜ˆÑÔ€AØÐØ	�TŒX�d‰^Œ^ÑˆØ€Hs   ÁAB-Â-B1Â4B1c                 óØ   — | d         }|j         |         }|                     d¦  «        }|€)t          j        t          j        |¦  «        dz   ¦  «        }dd|z  cxk    r|k     sn dS dS )Nr   Úwindow_lengthç      à?r   TF)ÚshapeÚgetr;   ÚfloorÚsqrt)ÚsamplesÚkwargsr&   ÚvaluesÚnr@   s         r   Ú"_differential_entropy_is_too_smallrJ   ¨   sx   € Ø�QŒZ€FØŒ�TÔ€AØ—J’J˜Ñ/Ô/€MØÐÝœ
¥4¤9¨Q¡<¤<°#Ñ#5Ñ6Ô6ˆØ��MÑ!Ð%Ð%Ò%Ð% AÒ%Ð%Ð%Ð%ØˆtØˆ5r   c                 ó   — | S r   r   r   s    r   r   r   µ   r   r   c                 ó   — | fS r   r   r   s     r   r   r   µ   s   € ¸A¸4€ r   )r    r!   r#   Úauto)r@   r%   r&   ÚmethodrH   r@   rN   c                ó  — t          | ¦  «        }t          | d|¬¦  «        } |                     | |d¦  «        } | j        d         }|€)t	          j        t	          j        |¦  «        dz   ¦  «        }dd|z  cxk    r|k     sn t          d|› d|› d	�¦  «        ‚|�|d
k    rt          d¦  «        ‚|                     | d¬¦  «        }t          t          t          t          t          dœ}|                     ¦   «         }||vr!dt          |¦  «        › �}	t          |	¦  «        ‚|dk    r|dk    rd}n|dk    rd}nd} ||         |||¬¦  «        }
|�|
t	          j        |¦  «        z  }
|                     |
| j        ¦  «        S )aV  Given a sample of a distribution, estimate the differential entropy.

    Several estimation methods are available using the `method` parameter. By
    default, a method is selected based the size of the sample.

    Parameters
    ----------
    values : sequence
        Sample from a continuous distribution.
    window_length : int, optional
        Window length for computing Vasicek estimate. Must be an integer
        between 1 and half of the sample size. If ``None`` (the default), it
        uses the heuristic value

        .. math::
            \left \lfloor \sqrt{n} + 0.5 \right \rfloor

        where :math:`n` is the sample size. This heuristic was originally
        proposed in [2]_ and has become common in the literature.
    base : float, optional
        The logarithmic base to use, defaults to ``e`` (natural logarithm).
    axis : int, optional
        The axis along which the differential entropy is calculated.
        Default is 0.
    method : {'vasicek', 'van es', 'ebrahimi', 'correa', 'auto'}, optional
        The method used to estimate the differential entropy from the sample.
        Default is ``'auto'``.  See Notes for more information.

    Returns
    -------
    entropy : float
        The calculated differential entropy.

    Notes
    -----
    This function will converge to the true differential entropy in the limit

    .. math::
        n \to \infty, \quad m \to \infty, \quad \frac{m}{n} \to 0

    The optimal choice of ``window_length`` for a given sample size depends on
    the (unknown) distribution. Typically, the smoother the density of the
    distribution, the larger the optimal value of ``window_length`` [1]_.

    The following options are available for the `method` parameter.

    * ``'vasicek'`` uses the estimator presented in [1]_. This is
      one of the first and most influential estimators of differential entropy.
    * ``'van es'`` uses the bias-corrected estimator presented in [3]_, which
      is not only consistent but, under some conditions, asymptotically normal.
    * ``'ebrahimi'`` uses an estimator presented in [4]_, which was shown
      in simulation to have smaller bias and mean squared error than
      the Vasicek estimator.
    * ``'correa'`` uses the estimator presented in [5]_ based on local linear
      regression. In a simulation study, it had consistently smaller mean
      square error than the Vasiceck estimator, but it is more expensive to
      compute.
    * ``'auto'`` selects the method automatically (default). Currently,
      this selects ``'van es'`` for very small samples (<10), ``'ebrahimi'``
      for moderate sample sizes (11-1000), and ``'vasicek'`` for larger
      samples, but this behavior is subject to change in future versions.

    All estimators are implemented as described in [6]_.

    References
    ----------
    .. [1] Vasicek, O. (1976). A test for normality based on sample entropy.
           Journal of the Royal Statistical Society:
           Series B (Methodological), 38(1), 54-59.
    .. [2] Crzcgorzewski, P., & Wirczorkowski, R. (1999). Entropy-based
           goodness-of-fit test for exponentiality. Communications in
           Statistics-Theory and Methods, 28(5), 1183-1202.
    .. [3] Van Es, B. (1992). Estimating functionals related to a density by a
           class of statistics based on spacings. Scandinavian Journal of
           Statistics, 61-72.
    .. [4] Ebrahimi, N., Pflughoeft, K., & Soofi, E. S. (1994). Two measures
           of sample entropy. Statistics & Probability Letters, 20(3), 225-234.
    .. [5] Correa, J. C. (1995). A new estimator of entropy. Communications
           in Statistics-Theory and Methods, 24(10), 2439-2449.
    .. [6] Noughabi, H. A. (2015). Entropy Estimation Using Numerical Methods.
           Annals of Data Science, 2(2), 231-241.
           https://link.springer.com/article/10.1007/s40745-015-0045-9

    Examples
    --------
    >>> import numpy as np
    >>> from scipy.stats import differential_entropy, norm

    Entropy of a standard normal distribution:

    >>> rng = np.random.default_rng()
    >>> values = rng.standard_normal(100)
    >>> differential_entropy(values)
    1.3407817436640392

    Compare with the true entropy:

    >>> float(norm.entropy())
    1.4189385332046727

    For several sample sizes between 5 and 1000, compare the accuracy of
    the ``'vasicek'``, ``'van es'``, and ``'ebrahimi'`` methods. Specifically,
    compare the root mean squared error (over 1000 trials) between the estimate
    and the true differential entropy of the distribution.

    >>> from scipy import stats
    >>> import matplotlib.pyplot as plt
    >>>
    >>>
    >>> def rmse(res, expected):
    ...     '''Root mean squared error'''
    ...     return np.sqrt(np.mean((res - expected)**2))
    >>>
    >>>
    >>> a, b = np.log10(5), np.log10(1000)
    >>> ns = np.round(np.logspace(a, b, 10)).astype(int)
    >>> reps = 1000  # number of repetitions for each sample size
    >>> expected = stats.expon.entropy()
    >>>
    >>> method_errors = {'vasicek': [], 'van es': [], 'ebrahimi': []}
    >>> for method in method_errors:
    ...     for n in ns:
    ...        rvs = stats.expon.rvs(size=(reps, n), random_state=rng)
    ...        res = stats.differential_entropy(rvs, method=method, axis=-1)
    ...        error = rmse(res, expected)
    ...        method_errors[method].append(error)
    >>>
    >>> for method, errors in method_errors.items():
    ...     plt.loglog(ns, errors, label=method)
    >>>
    >>> plt.legend()
    >>> plt.xlabel('sample size')
    >>> plt.ylabel('RMSE (1000 trials)')
    >>> plt.title('Entropy Estimator Error (Exponential Distribution)')

    T)Úforce_floatingr+   r   NrA   r   zWindow length (z7) must be positive and less than half the sample size (z).r   r)   r2   )Úvasicekúvan esÚcorreaÚebrahimirM   z`method` must be one of rM   é
   rR   iè  rT   rQ   r.   )r   r   ÚmoveaxisrB   r;   rD   rE   r3   ÚsortÚ_vasicek_entropyÚ_van_es_entropyÚ_correa_entropyÚ_ebrahimi_entropyÚlowerÚsetr<   ÚastypeÚdtype)rH   r@   r%   r&   rN   r+   rI   Úsorted_dataÚmethodsÚmessageÚress              r   r   r   ³   sÆ  € õj 
˜Ñ	 Ô	 €BÝ˜¨t¸Ð;Ñ;Ô;€FØ�[Š[˜  rÑ*Ô*€FØŒ�RÔ€AàÐÝœ
¥4¤9¨Q¡<¤<°#Ñ#5Ñ6Ô6ˆà��MÑ!Ð%Ð%Ò%Ð% AÒ%Ð%Ð%Ð%Ýð0˜mð 0ð 0Ø*+ð0ð 0ð 0ñ
ô 
ð 	
ð
 Ð˜D AšI˜IÝÐFÑGÔGÐGà—'’'˜& r�'Ñ*Ô*€Kå*Ý(Ý(Ý,Ý'ð	)ð )€Gð
 �\Š\‰^Œ^€FØ�WÐÐØ;­S°©\¬\Ð;Ð;ˆÝ˜Ñ!Ô!Ð!à�ÒÐØ�Š7ˆ7ØˆFˆFØ�$ŠYˆYØˆFˆFàˆFà
ˆ'�&Œ/˜+ }¸Ð
<Ñ
<Ô
<€CàÐØ�tŒx˜‰~Œ~Ñˆð �9Š9�S˜&œ,Ñ'Ô'Ð'r   c                óÜ   — | j         dd…         |fz   }|                     | ddd…f         |¦  «        }|                     | ddd…f         |¦  «        }|                     || |fd¬¦  «        S )z9Pad the data for computing the rolling window difference.Nr   .r   r2   )rB   Úbroadcast_toÚconcat)ÚXÚmr+   rB   ÚXlÚXrs         r   Ú_pad_along_last_axisrk   w  st   € ð ŒG�C�R�CŒL˜A˜4Ñ€EØ	�Š˜˜3   ˜7œ UÑ	+Ô	+€BØ	�Š˜˜3   ˜8œ eÑ	,Ô	,€BØ�9Š9�b˜!˜R�[ rˆ9Ñ*Ô*Ð*r   c                óè   — | j         d         }t          | ||¬¦  «        } | dd|z  d…f         | ddd|z  …f         z
  }|                     |d|z  z  |z  ¦  «        }|                     |d¬¦  «        S )z:Compute the Vasicek estimator as described in [6] Eq. 1.3.r   r.   .r   Néþÿÿÿr2   )rB   rk   r<   Úmean)rg   rh   r+   rI   ÚdifferencesÚlogss         r   rX   rX   €  s€   € à	Œ�Œ€AÝ˜Q  bÐ)Ñ)Ô)€AØ�C˜˜Q™˜˜�K”. 1 S¨)¨B°©F¨) ^Ô#4Ñ4€KØ�6Š6�!�Q�q‘S‘'˜KÑ'Ñ(Ô(€DØ�7Š7�4˜bˆ7Ñ!Ô!Ð!r   c                ó´  — | j         d         }| d|d…f         | dd| …f         z
  }d||z
  z  |                     |                     |dz   |z  |z  ¦  «        d¬¦  «        z  }|                     ||dz   |j        t          | ¦  «        ¬¦  «        }||                     d|z  ¦  «        z   t          j        |¦  «        z   t          j        |dz   ¦  «        z
  S )z1Compute the van Es estimator as described in [6].r   .Nr   r2   ©r_   Údevice)rB   r6   r<   Úaranger_   r   r;   )rg   rh   r+   rI   Ú
differenceÚterm1Úks          r   rY   rY   ‰  sÊ   € ð 	
Œ�Œ€AØ�3˜˜˜�7”˜a  S q b S œkÑ)€JØˆq�‰s‰G�b—f’f˜RŸVšV Q q¡S¨!¡G¨jÑ$8Ñ9Ô9À�fÑCÔCÑC€EØ
�	Š	�!�Q�q‘S ¤µI¸a±L´Lˆ	ÑAÔA€AØ�2—6’6˜!˜A™#‘;”;Ñ¥¤¨!¡¤Ñ,­t¬x¸¸!¹©}¬}Ñ<Ð<r   c                óà  — | j         d         }t          | ||¬¦  «        } | dd|z  d…f         | ddd|z  …f         z
  }|                     d|dz   | j        t	          | ¦  «        ¬¦  «        }|                     ||k    d|dz
  |z  z   d	¦  «        }|                     |||z
  dz   k    d||z
  |z  z   |¦  «        }|                     ||z  ||z  z  ¦  «        }|                     |d¬
¦  «        S )z3Compute the Ebrahimi estimator as described in [6].r   r.   .r   Nrm   r   rr   g       @r2   )rB   rk   rt   r_   r   Úwherer<   rn   )rg   rh   r+   rI   ro   ÚiÚcirp   s           r   r[   r[   ”  sù   € ð 	
Œ�Œ€AÝ˜Q  bÐ)Ñ)Ô)€Aà�C˜˜Q™˜˜�K”. 1 S¨)¨B°©F¨) ^Ô#4Ñ4€Kà
�	Š	�!�Q�q‘S ¤µ	¸!±´ˆ	Ñ=Ô=€AØ	�Š�!�q’&˜!˜q 1™u a™i™-¨Ñ	,Ô	,€BØ	�Š�!�q˜1‘u˜q‘y’. ! q¨1¡u¨a¡i¡-°Ñ	4Ô	4€Bà�6Š6�!�k‘/ R¨!¡VÑ,Ñ-Ô-€DØ�7Š7�4˜bˆ7Ñ!Ô!Ð!r   c                ó4  — | j         d         }t          | ||¬¦  «        } |                     d|dz   t          | ¦  «        ¬¦  «        }|                     | |dz   t          | ¦  «        ¬¦  «        dd…df         }||z   }||z   dz
  }|                     | d|f         dd¬	¦  «        }| d|f         |z
  }	|                     |	|z  d¬
¦  «        }
||                     |	dz  d¬
¦  «        z  }|                     |                     |
|z  ¦  «        d¬
¦  «         S )z1Compute the Correa estimator as described in [6].r   r.   r   )rs   N.rm   Tr/   r2   r   )rB   rk   rt   r   rn   r6   r<   )rg   rh   r+   rI   rz   ÚdjÚjÚj0ÚXibarru   ÚnumÚdens               r   rZ   rZ   ¤  s  € ð 	
Œ�Œ€AÝ˜Q  bÐ)Ñ)Ô)€Aà
�	Š	�!�Q�q‘S¥¨1¡¤ˆ	Ñ.Ô.€AØ	�Š�A�2�q˜‘s¥9¨Q¡<¤<ˆÑ	0Ô	0°°°°D°Ô	9€BØ	ˆB‰€AØ	
ˆQ‰�‰€Bà�GŠG�A�c˜2�g”J R°$ˆGÑ7Ô7€EØ�3˜�7”˜eÑ#€JØ
�&Š&�˜B‘ Rˆ&Ñ
(Ô
(€CØ
ˆB�FŠF�:˜q‘= rˆFÑ*Ô*Ñ
*€CØ�GŠG�B—F’F˜3˜s™7‘O”O¨"ˆGÑ-Ô-Ð-Ð-r   )NNr   )r   )Ú__doc__r;   Únumpyr4   Úscipyr   Ú_axis_nan_policyr   Úscipy._lib._array_apir   r   r   r	   r
   r   Ú__all__ÚtypingÚ	ArrayLikeÚfloatÚintÚnumberÚndarrayr   rJ   Ústrr   rk   rX   rY   r[   rZ   r   r   r   ú<module>r�      s¶  ððð ð €€€Ø Ð Ð Ð Ø Ð Ð Ð Ð Ð Ø 6Ð 6Ð 6Ð 6Ð 6Ð 6ðMð Mð Mð Mð Mð Mð Mð Mð Mð Mð Mð Mð Mð Mð Mð Mð Ð,Ð
-€ð €ÑÔØÐØ€Kðð ð Ð!2Ð!2¸4Øðñ ô ð .2Ø!%ØðJð J�”	Ô#ð JØ”	Ô# dÑ*ðJà˜$‘,ðJð ðJð ”˜RœZÑ'ð	Jð Jð Jñô ñ ÔðJðZð ð ð ð €ÑÔØÐØ€K˜1Ð.?Ð.?Ø0ðñ ô ð !%ØØØð|(ð |(ð |(ØŒIÔð|(ð ˜‘:ð|(ð �$‰,ð	|(ð
 ð|(ð ð|(ð „Y�”Ñð|(ð |(ð |(ñ	ô ñ Ôð
|(ð~+ð +ð +ð"ð "ð "ð=ð =ð =ð"ð "ð "ð .ð .ð .ð .ð .r   