o
    Ö­jî?  ã                   @   sÌ   d Z g d¢ZddlZddlmZmZ ddlmZ ddlmZ ddl	m
Z ddlmZmZmZmZ d#dd„Zd$dd„Zd%dd„Z		d&dd„Zd%dd„Zd'dd„Zd(dd„Zd)dd„Zd)dd „Zd)d!d"„ZdS )*zB
Additional statistics functions with support for masked arrays.

)
Úcompare_medians_msÚhdquantilesÚhdmedianÚhdquantiles_sdÚidealfourthsÚmedian_cihsÚmjciÚmquantiles_cimjÚrshÚtrimmed_mean_cié    N)Úfloat64Úndarray)ÚMaskedArrayé   )Ú_mstats_basic)ÚnormÚbetaÚtÚbinom©g      Ð?ç      à?g      è?Fc                 C   s€   dd„ }t j| dtd�} t t |¡¡}|du s| jdkr$|| ||ƒ}n| jdkr0td| j ƒ‚t  ||| ||¡}t j	|dd	�S )
a$  
    Computes quantile estimates with the Harrell-Davis method.

    The quantile estimates are calculated as a weighted linear combination
    of order statistics.

    Parameters
    ----------
    data : array_like
        Data array.
    prob : sequence, optional
        Sequence of probabilities at which to compute the quantiles.
    axis : int or None, optional
        Axis along which to compute the quantiles. If None, use a flattened
        array.
    var : bool, optional
        Whether to return the variance of the estimate.

    Returns
    -------
    hdquantiles : MaskedArray
        A (p,) array of quantiles (if `var` is False), or a (2,p) array of
        quantiles and variances (if `var` is True), where ``p`` is the
        number of quantiles.

    See Also
    --------
    hdquantiles_sd

    Examples
    --------
    >>> import numpy as np
    >>> from scipy.stats.mstats import hdquantiles
    >>>
    >>> # Sample data
    >>> data = np.array([1.2, 2.5, 3.7, 4.0, 5.1, 6.3, 7.0, 8.2, 9.4])
    >>>
    >>> # Probabilities at which to compute quantiles
    >>> probabilities = [0.25, 0.5, 0.75]
    >>>
    >>> # Compute Harrell-Davis quantile estimates
    >>> quantile_estimates = hdquantiles(data, prob=probabilities)
    >>>
    >>> # Display the quantile estimates
    >>> for i, quantile in enumerate(probabilities):
    ...     print(f"{int(quantile * 100)}th percentile: {quantile_estimates[i]}")
    25th percentile: 3.1505820231763066 # may vary
    50th percentile: 5.194344084883956
    75th percentile: 7.430626414674935

    c                 S   sH  t  t  |  ¡  t¡¡¡}|j}t  dt|ƒft	¡}|dk r*t j
|_|r&|S |d S t  |d ¡t|ƒ }tj}t|ƒD ]:\}}	|||d |	 |d d|	  ƒ}
|
dd… |
dd…  }t  ||¡}||d|f< t  ||| d ¡|d|f< q<|d |d|dkf< |d |d|dkf< |r t j
 |d|dkf< |d|dkf< |S |d S )zGComputes the HD quantiles for a 1D array. Returns nan for invalid data.é   r   r   Néÿÿÿÿ)ÚnpÚsqueezeÚsortÚ
compressedÚviewr   ÚsizeÚemptyÚlenr   ÚnanÚflatÚarangeÚfloatr   ÚcdfÚ	enumerateÚdot)ÚdataÚprobÚvarÚxsortedÚnÚhdÚvÚbetacdfÚiÚpÚ_wÚwÚhd_mean© r5   úW/var/www/html/CropPilot/venv/lib/python3.10/site-packages/scipy/stats/_mstats_extras.pyÚ_hd_1DP   s,    "zhdquantiles.<locals>._hd_1DF©ÚcopyÚdtypeNr   r   úDArray 'data' must be at most two dimensional, but got data.ndim = %d©r9   )
ÚmaÚarrayr   r   Ú
atleast_1dÚasarrayÚndimÚ
ValueErrorÚapply_along_axisÚfix_invalid)r(   r)   Úaxisr*   r7   r1   Úresultr5   r5   r6   r      s   4
ÿr   r   c                 C   s   t | dg||d�}| ¡ S )a9  
    Returns the Harrell-Davis estimate of the median along the given axis.

    Parameters
    ----------
    data : ndarray
        Data array.
    axis : int, optional
        Axis along which to compute the quantiles. If None, use a flattened
        array.
    var : bool, optional
        Whether to return the variance of the estimate.

    Returns
    -------
    hdmedian : MaskedArray
        The median values.  If ``var=True``, the variance is returned inside
        the masked array.  E.g. for a 1-D array the shape change from (1,) to
        (2,).

    r   )rE   r*   )r   r   )r(   rE   r*   rF   r5   r5   r6   r   |   s   r   c                 C   sv   dd„ }t j| dtd�} t t |¡¡}|du r|| |ƒ}n| jdkr*td| j ƒ‚t  ||| |¡}t j	|dd� 
¡ S )	aý  
    The standard error of the Harrell-Davis quantile estimates by jackknife.

    Parameters
    ----------
    data : array_like
        Data array.
    prob : sequence, optional
        Sequence of quantiles to compute.
    axis : int, optional
        Axis along which to compute the quantiles. If None, use a flattened
        array.

    Returns
    -------
    hdquantiles_sd : MaskedArray
        Standard error of the Harrell-Davis quantile estimates.

    See Also
    --------
    hdquantiles

    c                 S   s  t  |  ¡ ¡}t|ƒ}t  t|ƒt¡}|dk rt j|_t  |¡t	|d ƒ }t
j}t|ƒD ][\}}|||| |d|  ƒ}	|	dd… |	dd…  }
t  |¡}t  |
|dd…  ¡|dd…< |dd…  t  |
ddd… |ddd…  ¡ddd… 7  < t  | ¡ |d  ¡||< q-|S )z%Computes the std error for 1D arrays.r   r   Nr   r   )r   r   r   r    r   r   r!   r"   r#   r$   r   r%   r&   Ú
zeros_likeÚcumsumÚsqrtr*   )r(   r)   r+   r,   ÚhdsdÚvvr/   r0   r1   r2   r3   Úmx_r5   r5   r6   Ú_hdsd_1D®   s   
<z hdquantiles_sd.<locals>._hdsd_1DFr8   Nr   r;   r<   )r=   r>   r   r   r?   r@   rA   rB   rC   rD   Úravel)r(   r)   rE   rM   r1   rF   r5   r5   r6   r   –   s   
ÿr   ©çš™™™™™É?rP   ©TTçš™™™™™©?c           
      C   s|   t j| dd�} tj| |||d�}| |¡}tj| |||d�}| |¡d }t d|d  |¡}	t	 ||	|  ||	|  f¡S )a³  
    Selected confidence interval of the trimmed mean along the given axis.

    Parameters
    ----------
    data : array_like
        Input data.
    limits : {None, tuple}, optional
        None or a two item tuple.
        Tuple of the percentages to cut on each side of the array, with respect
        to the number of unmasked data, as floats between 0. and 1. If ``n``
        is the number of unmasked data before trimming, then
        (``n * limits[0]``)th smallest data and (``n * limits[1]``)th
        largest data are masked.  The total number of unmasked data after
        trimming is ``n * (1. - sum(limits))``.
        The value of one limit can be set to None to indicate an open interval.

        Defaults to (0.2, 0.2).
    inclusive : (2,) tuple of boolean, optional
        If relative==False, tuple indicating whether values exactly equal to
        the absolute limits are allowed.
        If relative==True, tuple indicating whether the number of data being
        masked on each side should be rounded (True) or truncated (False).

        Defaults to (True, True).
    alpha : float, optional
        Confidence level of the intervals.

        Defaults to 0.05.
    axis : int, optional
        Axis along which to cut. If None, uses a flattened version of `data`.

        Defaults to None.

    Returns
    -------
    trimmed_mean_ci : (2,) ndarray
        The lower and upper confidence intervals of the trimmed data.

    Fr<   )ÚlimitsÚ	inclusiverE   r   ç       @)
r=   r>   ÚmstatsÚtrimrÚmeanÚtrimmed_stdeÚcountr   Úppfr   )
r(   rS   rT   ÚalpharE   ÚtrimmedÚtmeanÚtstdeÚdfÚtppfr5   r5   r6   r
   Õ   s   *
r
   c                 C   s`   dd„ }t j| dd�} | jdkrtd| j ƒ‚t t |¡¡}|du r(|| |ƒS t  ||| |¡S )a„  
    Returns the Maritz-Jarrett estimators of the standard error of selected
    experimental quantiles of the data.

    Parameters
    ----------
    data : ndarray
        Data array.
    prob : sequence, optional
        Sequence of quantiles to compute.
    axis : int or None, optional
        Axis along which to compute the quantiles. If None, use a flattened
        array.

    c                 S   sÖ   t  |  ¡ ¡} | j}t  |¡| d  t¡}tj}t  	t
|ƒt¡}t jd|d td�| }|d|  }t|ƒD ]1\}}	|||	d ||	 ƒ|||	d ||	 ƒ }
t  |
| ¡}t  |
| d ¡}t  ||d  ¡||< q7|S )Nr   r   )r:   g      ð?r   )r   r   r   r   r>   ÚastypeÚintr   r%   r   r    r   r#   r&   r'   rI   )r(   r1   r,   r)   r/   ÚmjÚxÚyr0   ÚmÚWÚC1ÚC2r5   r5   r6   Ú_mjci_1D  s   (zmjci.<locals>._mjci_1DFr<   r   r;   N)r=   r>   rA   rB   r   r?   r@   rC   )r(   r)   rE   rk   r1   r5   r5   r6   r     s   
ÿ
r   c                 C   sZ   t |d| ƒ}t d|d  ¡}tj| |dd|d�}t| ||d�}|||  |||  fS )aÕ  
    Computes the alpha confidence interval for the selected quantiles of the
    data, with Maritz-Jarrett estimators.

    Parameters
    ----------
    data : ndarray
        Data array.
    prob : sequence, optional
        Sequence of quantiles to compute.
    alpha : float, optional
        Confidence level of the intervals.
    axis : int or None, optional
        Axis along which to compute the quantiles.
        If None, use a flattened array.

    Returns
    -------
    ci_lower : ndarray
        The lower boundaries of the confidence interval.  Of the same length as
        `prob`.
    ci_upper : ndarray
        The upper boundaries of the confidence interval.  Of the same length as
        `prob`.

    r   rU   r   )ÚalphapÚbetaprE   ©rE   )Úminr   r[   rV   Ú
mquantilesr   )r(   r)   r\   rE   ÚzÚxqÚsmjr5   r5   r6   r   5  s
   r   c                 C   sX   dd„ }t j| dd�} |du r|| |ƒ}|S | jdkr"td| j ƒ‚t  ||| |¡}|S )aA  
    Computes the alpha-level confidence interval for the median of the data.

    Uses the Hettmasperger-Sheather method.

    Parameters
    ----------
    data : array_like
        Input data. Masked values are discarded. The input should be 1D only,
        or `axis` should be set to None.
    alpha : float, optional
        Confidence level of the intervals.
    axis : int or None, optional
        Axis along which to compute the quantiles. If None, use a flattened
        array.

    Returns
    -------
    median_cihs
        Alpha level confidence interval.

    c           	      S   s>  t  |  ¡ ¡} t| ƒ}t|d| ƒ}tt |d |d¡ƒ}t || |d¡t |d |d¡ }|d| k rK|d8 }t || |d¡t |d |d¡ }t || d |d¡t ||d¡ }|d | ||  }|| | t	||d|  |  ƒ }|| |  d| | |d    || || d   d| | ||    f}|S )Nr   rU   r   r   )
r   r   r   r    ro   rc   r   Ú_ppfr%   r$   )	r(   r\   r,   ÚkÚgkÚgkkÚIÚlambdÚlimsr5   r5   r6   Ú_cihs_1Dn  s   $$$$&ÿzmedian_cihs.<locals>._cihs_1DFr<   Nr   r;   )r=   r>   rA   rB   rC   )r(   r\   rE   r{   rF   r5   r5   r6   r   W  s   

ûÿr   c                 C   sn   t j| |d�t j||d�}}tj| |d�tj||d�}}t || ¡t  |d |d  ¡ }dt |¡ S )a"  
    Compares the medians from two independent groups along the given axis.

    The comparison is performed using the McKean-Schrader estimate of the
    standard error of the medians.

    Parameters
    ----------
    group_1 : array_like
        First dataset.  Has to be of size >=7.
    group_2 : array_like
        Second dataset.  Has to be of size >=7.
    axis : int, optional
        Axis along which the medians are estimated. If None, the arrays are
        flattened.  If `axis` is not None, then `group_1` and `group_2`
        should have the same shape.

    Returns
    -------
    compare_medians_ms : {float, ndarray}
        If `axis` is None, then returns a float, otherwise returns a 1-D
        ndarray of floats with a length equal to the length of `group_1`
        along `axis`.

    Examples
    --------

    >>> from scipy import stats
    >>> a = [1, 2, 3, 4, 5, 6, 7]
    >>> b = [8, 9, 10, 11, 12, 13, 14]
    >>> stats.mstats.compare_medians_ms(a, b, axis=None)
    1.0693225866553746e-05

    The function is vectorized to compute along a given axis.

    >>> import numpy as np
    >>> rng = np.random.default_rng()
    >>> x = rng.random(size=(3, 7))
    >>> y = rng.random(size=(3, 8))
    >>> stats.mstats.compare_medians_ms(x, y, axis=1)
    array([0.36908985, 0.36092538, 0.2765313 ])

    References
    ----------
    .. [1] McKean, Joseph W., and Ronald M. Schrader. "A comparison of methods
       for studentizing the sample median." Communications in
       Statistics-Simulation and Computation 13.6 (1984): 751-773.

    rn   r   r   )	r=   ÚmedianrV   Ústde_medianr   ÚabsrI   r   r%   )Úgroup_1Úgroup_2rE   Úmed_1Úmed_2Ústd_1Ústd_2rh   r5   r5   r6   r   Š  s   2ÿ$r   c                 C   s:   dd„ }t j| |d� t¡} |du r|| ƒS t  ||| ¡S )aC  
    Returns an estimate of the lower and upper quartiles.

    Uses the ideal fourths algorithm.

    Parameters
    ----------
    data : array_like
        Input array.
    axis : int, optional
        Axis along which the quartiles are estimated. If None, the arrays are
        flattened.

    Returns
    -------
    idealfourths : {list of floats, masked array}
        Returns the two internal values that divide `data` into four parts
        using the ideal fourths algorithm either along the flattened array
        (if `axis` is None) or along `axis` of `data`.

    c                 S   s’   |   ¡ }t|ƒ}|dk rtjtjgS t|d d dƒ\}}t|ƒ}d| ||d   |||   }|| }d| ||  |||d    }||gS )Né   g      @g«ªªªªªÚ?r   )r   r    r   r!   Údivmodrc   )r(   re   r,   ÚjÚhÚqloru   Úqupr5   r5   r6   Ú_idfÙ  s     zidealfourths.<locals>._idfrn   N)r=   r   r   r   rC   )r(   rE   r‹   r5   r5   r6   r   Ã  s
   r   c                 C   sÖ   t j| dd�} |du r| }nt t |¡¡}| jdkrtdƒ‚|  ¡ }t| dd�}d|d |d	   |d
  }| dd…df |ddd…f | k 	d	¡}| dd…df |ddd…f | k  	d	¡}|| d| |  S )aé  
    Evaluates Rosenblatt's shifted histogram estimators for each data point.

    Rosenblatt's estimator is a centered finite-difference approximation to the
    derivative of the empirical cumulative distribution function.

    Parameters
    ----------
    data : sequence
        Input data, should be 1-D. Masked values are ignored.
    points : sequence or None, optional
        Sequence of points where to evaluate Rosenblatt shifted histogram.
        If None, use the data.

    Fr<   Nr   z#The input array should be 1D only !rn   g333333ó?r   r   rP   rU   )
r=   r>   r   r?   r@   rA   ÚAttributeErrorrZ   r   Úsum)r(   Úpointsr,   Úrrˆ   ÚnhiÚnlor5   r5   r6   r	   ë  s   
**r	   )r   NF)r   F)r   N)rO   rQ   rR   N)r   rR   N)rR   N)N)Ú__doc__Ú__all__Únumpyr   r   r   Únumpy.mar=   r   Ú r   rV   Úscipy.stats.distributionsr   r   r   r   r   r   r   r
   r   r   r   r   r   r	   r5   r5   r5   r6   Ú<module>   s(    

`
?
ÿ
3
-
"
3
9(