§
    rŠtjq=  ã                   ó.  — d Z ddlZddlmZ ddlmZmZ ddlZddl	m
Z
 ddlmZ ddlmZ ddlmZ dd	lmZmZ dd
lmZ ddlmZ ddlmZ ddlmZ ddlmZmZ ddlm Z   ej!        ej"        ¦  «        j#        Z$d„ Z%dd„Z&d„ Z'd„ Z( G d„ dee¦  «        Z)dS )z<
A Theil-Sen Estimator for Multiple Linear Regression Model
é    N)Úcombinations)ÚIntegralÚReal)Úeffective_n_jobs)Úlinalg)Úget_lapack_funcs)Úbinom)ÚRegressorMixinÚ_fit_context)ÚConvergenceWarning)ÚLinearModel)Úcheck_random_state)ÚInterval)ÚParallelÚdelayed)Úvalidate_datac                 óp  — | |z
  }t          j        t          j        |dz  d¬¦  «        ¦  «        }|t          k    }t	          |                     ¦   «         | j        d         k     ¦  «        }||         }||         dd…t           j        f         }t          j        t          j        ||z  d¬¦  «        ¦  «        }|t          k    r>t          j        | |dd…f         |z  d¬¦  «        t          j        d|z  d¬¦  «        z  }nd}d}t          dd||z  z
  ¦  «        |z  t          d||z  ¦  «        |z  z   S )u	  Modified Weiszfeld step.

    This function defines one iteration step in order to approximate the
    spatial median (L1 median). It is a form of an iteratively re-weighted
    least squares method.

    Parameters
    ----------
    X : array-like of shape (n_samples, n_features)
        Training vector, where `n_samples` is the number of samples and
        `n_features` is the number of features.

    x_old : ndarray of shape = (n_features,)
        Current start vector.

    Returns
    -------
    x_new : ndarray of shape (n_features,)
        New iteration step.

    References
    ----------
    - On Computation of Spatial Median for Robust Data Mining, 2005
      T. KÃ¤rkkÃ¤inen and S. Ã„yrÃ¤mÃ¶
      http://users.jyu.fi/~samiayr/pdf/ayramo_eurogen05.pdf
    é   é   ©Úaxisr   Ng      ð?ç        )ÚnpÚsqrtÚsumÚ_EPSILONÚintÚshapeÚnewaxisr   ÚnormÚmaxÚmin)ÚXÚx_oldÚdiffÚ	diff_normÚmaskÚis_x_old_in_XÚquotient_normÚnew_directions           ú]/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/sklearn/linear_model/_theil_sen.pyÚ_modified_weiszfeld_stepr,      s>  € ð6 ˆu‰9€DÝ”�œ˜t Q™w¨QÐ/Ñ/Ô/Ñ0Ô0€IØ�Ò €Då˜Ÿš™
œ
 Q¤W¨Q¤ZÒ/Ñ0Ô0€Mà�Œ:€DØ˜$”   ¥2¤: Ô.€IÝ”K¥¤ t¨iÑ'7¸aÐ @Ñ @Ô @ÑAÔA€Mà•xÒÐÝœ˜q  q q q œz¨IÑ5¸AÐ>Ñ>Ô>ÅÄØ�	‰M ðB
ñ B
ô B
ñ 
ˆˆð ˆØˆõ 	ˆC��} }Ñ4Ñ4Ñ5Ô5¸ÑEÝ
ˆc�= =Ñ0Ñ
1Ô
1°EÑ
9ñ	:ðó    é,  çü©ñÒMbP?c                 óš  — | j         d         dk    r*dt          j        |                      ¦   «         d¬¦  «        fS |dz  }t          j        | d¬¦  «        }t          |¦  «        D ]4}t          | |¦  «        }t          j        ||z
  dz  ¦  «        |k     r n1|}Œ5t          j	        d 
                    |¬¦  «        t          ¦  «         ||fS )	u	  Spatial median (L1 median).

    The spatial median is member of a class of so-called M-estimators which
    are defined by an optimization problem. Given a number of p points in an
    n-dimensional space, the point x minimizing the sum of all distances to the
    p other points is called spatial median.

    Parameters
    ----------
    X : array-like of shape (n_samples, n_features)
        Training vector, where `n_samples` is the number of samples and
        `n_features` is the number of features.

    max_iter : int, default=300
        Maximum number of iterations.

    tol : float, default=1.e-3
        Stop the algorithm if spatial_median has converged.

    Returns
    -------
    spatial_median : ndarray of shape = (n_features,)
        Spatial median.

    n_iter : int
        Number of iterations needed.

    References
    ----------
    - On Computation of Spatial Median for Robust Data Mining, 2005
      T. KÃ¤rkkÃ¤inen and S. Ã„yrÃ¤mÃ¶
      http://users.jyu.fi/~samiayr/pdf/ayramo_eurogen05.pdf
    r   T)Úkeepdimsr   r   r   zYMaximum number of iterations {max_iter} reached in spatial median for TheilSen regressor.)Úmax_iter)r   r   ÚmedianÚravelÚmeanÚranger,   r   ÚwarningsÚwarnÚformatr   )r#   r2   ÚtolÚspatial_median_oldÚn_iterÚspatial_medians         r+   Ú_spatial_medianr>   P   sß   € ðD 	„wˆq„z�Q‚€Ø•"”)˜AŸGšG™IœI°Ð5Ñ5Ô5Ð5Ð5àˆA�I€CÝœ ¨Ð+Ñ+Ô+Ðå˜‘/”/ð 
ð 
ˆÝ1°!Ð5GÑHÔHˆÝŒ6Ð%¨Ñ6¸1Ñ<Ñ=Ô=ÀÒCÐCØˆEà!/ÐÐåŒðçŠv˜xˆvÑ(Ô(Ýñ		
ô 	
ð 	
ð �>Ð!Ð!r-   c                 ó<   — ddd|z  z  | |z
  dz   z  |z   dz
  | z  z
  S )a  Approximation of the breakdown point.

    Parameters
    ----------
    n_samples : int
        Number of samples.

    n_subsamples : int
        Number of subsamples to consider.

    Returns
    -------
    breakdown_point : float
        Approximation of breakdown point.
    r   g      à?© )Ú	n_samplesÚn_subsampless     r+   Ú_breakdown_pointrC   ˆ   sG   € ð" 	
à�A˜Ñ$Ñ%¨°\Ñ)AÀAÑ)EÑFØñàñð ññ	ðr-   c                 óà  — t          |¦  «        }| j        d         |z   }|j        d         }t          j        |j        d         |f¦  «        }t          j        ||f¦  «        }t          j        t          ||¦  «        ¦  «        }t          d||f¦  «        \  }	t          |¦  «        D ]D\  }
}| |dd…f         |dd…|d…f<   ||         |d|…<    |	||¦  «        d         d|…         ||
<   ŒE|S )a�  Least Squares Estimator for TheilSenRegressor class.

    This function calculates the least squares method on a subset of rows of X
    and y defined by the indices array. Optionally, an intercept column is
    added if intercept is set to true.

    Parameters
    ----------
    X : array-like of shape (n_samples, n_features)
        Design matrix, where `n_samples` is the number of samples and
        `n_features` is the number of features.

    y : ndarray of shape (n_samples,)
        Target vector, where `n_samples` is the number of samples.

    indices : ndarray of shape (n_subpopulation, n_subsamples)
        Indices of all subsamples with respect to the chosen subpopulation.

    fit_intercept : bool
        Fit intercept or not.

    Returns
    -------
    weights : ndarray of shape (n_subpopulation, n_features + intercept)
        Solution matrix of n_subpopulation solved least square problems.
    r   r   )ÚgelssN)	r   r   r   ÚemptyÚonesÚzerosr!   r   Ú	enumerate)r#   ÚyÚindicesÚfit_interceptÚ
n_featuresrB   ÚweightsÚX_subpopulationÚy_subpopulationÚlstsqÚindexÚsubsets               r+   Ú_lstsqrT   £   s  € õ6 ˜Ñ&Ô&€MØ”˜”˜mÑ+€JØ”= Ô#€LÝŒh˜œ aÔ(¨*Ð5Ñ6Ô6€GÝ”g˜|¨ZÐ8Ñ9Ô9€Oå”h¥ L°*Ñ =Ô =Ñ?Ô?€OÝ 
¨_¸oÐ,NÑOÔO�H€Uå" 7Ñ+Ô+ð Qð Q‰ˆˆvØ-.¨v°q°q°q¨y¬\ˆ˜˜˜˜=˜>˜>Ð)Ñ*Ø)*¨6¬ˆ˜˜˜Ñ&Ø˜˜°Ñ@Ô@ÀÔCÀKÀZÀKÔPˆ�‰ˆà€Nr-   c            
       óà   — e Zd ZU dZdg eeddd¬¦  «        gdeg eeddd¬¦  «        g eeddd¬¦  «        gd	gdegd
gdœZee	d<   dddddddddœd„Z
d„ Z ed¬¦  «        d„ ¦   «         ZdS )ÚTheilSenRegressoraë  Theil-Sen Estimator: robust multivariate regression model.

    The algorithm calculates least square solutions on subsets with size
    n_subsamples of the samples in X. Any value of n_subsamples between the
    number of features and samples leads to an estimator with a compromise
    between robustness and efficiency. Since the number of least square
    solutions is "n_samples choose n_subsamples", it can be extremely large
    and can therefore be limited with max_subpopulation. If this limit is
    reached, the subsets are chosen randomly. In a final step, the spatial
    median (or L1 median) is calculated of all least square solutions.

    Read more in the :ref:`User Guide <theil_sen_regression>`.

    Parameters
    ----------
    fit_intercept : bool, default=True
        Whether to calculate the intercept for this model. If set
        to false, no intercept will be used in calculations.

    max_subpopulation : int, default=1e4
        Instead of computing with a set of cardinality 'n choose k', where n is
        the number of samples and k is the number of subsamples (at least
        number of features), consider only a stochastic subpopulation of a
        given maximal size if 'n choose k' is larger than max_subpopulation.
        For other than small problem sizes this parameter will determine
        memory usage and runtime if n_subsamples is not changed. Note that the
        data type should be int but floats such as 1e4 can be accepted too.

    n_subsamples : int, default=None
        Number of samples to calculate the parameters. This is at least the
        number of features (plus 1 if fit_intercept=True) and the number of
        samples as a maximum. A lower number leads to a higher breakdown
        point and a low efficiency while a high number leads to a low
        breakdown point and a high efficiency. If None, take the
        minimum number of subsamples leading to maximal robustness.
        If n_subsamples is set to n_samples, Theil-Sen is identical to least
        squares.

    max_iter : int, default=300
        Maximum number of iterations for the calculation of spatial median.

    tol : float, default=1e-3
        Tolerance when calculating spatial median.

    random_state : int, RandomState instance or None, default=None
        A random number generator instance to define the state of the random
        permutations generator. Pass an int for reproducible output across
        multiple function calls.
        See :term:`Glossary <random_state>`.

    n_jobs : int, default=None
        Number of CPUs to use during the cross validation.
        ``None`` means 1 unless in a :obj:`joblib.parallel_backend` context.
        ``-1`` means using all processors. See :term:`Glossary <n_jobs>`
        for more details.

    verbose : bool, default=False
        Verbose mode when fitting the model.

    Attributes
    ----------
    coef_ : ndarray of shape (n_features,)
        Coefficients of the regression model (median of distribution).

    intercept_ : float
        Estimated intercept of regression model.

    breakdown_ : float
        Approximated breakdown point.

    n_iter_ : int
        Number of iterations needed for the spatial median.

    n_subpopulation_ : int
        Number of combinations taken into account from 'n choose k', where n is
        the number of samples and k is the number of subsamples.

    n_features_in_ : int
        Number of features seen during :term:`fit`.

        .. versionadded:: 0.24

    feature_names_in_ : ndarray of shape (`n_features_in_`,)
        Names of features seen during :term:`fit`. Defined only when `X`
        has feature names that are all strings.

        .. versionadded:: 1.0

    See Also
    --------
    HuberRegressor : Linear regression model that is robust to outliers.
    RANSACRegressor : RANSAC (RANdom SAmple Consensus) algorithm.
    SGDRegressor : Fitted by minimizing a regularized empirical loss with SGD.

    References
    ----------
    - Theil-Sen Estimators in a Multiple Linear Regression Model, 2009
      Xin Dang, Hanxiang Peng, Xueqin Wang and Heping Zhang
      http://home.olemiss.edu/~xdang/papers/MTSE.pdf

    Examples
    --------
    >>> from sklearn.linear_model import TheilSenRegressor
    >>> from sklearn.datasets import make_regression
    >>> X, y = make_regression(
    ...     n_samples=200, n_features=2, noise=4.0, random_state=0)
    >>> reg = TheilSenRegressor(random_state=0).fit(X, y)
    >>> reg.score(X, y)
    0.9884
    >>> reg.predict(X[:1,])
    array([-31.5871])
    Úbooleanr   NÚleft)Úclosedr   r   Úrandom_stateÚverbose©rL   Úmax_subpopulationrB   r2   r:   rZ   Ún_jobsr[   Ú_parameter_constraintsTg     ˆÃ@r.   r/   Fc                óv   — || _         || _        || _        || _        || _        || _        || _        || _        d S ©Nr\   )	ÚselfrL   r]   rB   r2   r:   rZ   r^   r[   s	            r+   Ú__init__zTheilSenRegressor.__init__M  sD   € ð +ˆÔØ!2ˆÔØ(ˆÔØ ˆŒØˆŒØ(ˆÔØˆŒØˆŒˆˆr-   c           	      ó  — | j         }| j        r|dz   }n|}|��||k    r#t          d                     ||¦  «        ¦  «        ‚||k    r6||k    r/| j        rdnd}t          d                     |||¦  «        ¦  «        ‚n:||k    r#t          d                     ||¦  «        ¦  «        ‚nt	          ||¦  «        }t          dt          j        t          ||¦  «        ¦  «        ¦  «        }t          t	          | j
        |¦  «        ¦  «        }||fS )Nr   z=Invalid parameter since n_subsamples > n_samples ({0} > {1}).z+1Ú zAInvalid parameter since n_features{0} > n_subsamples ({1} > {2}).z\Invalid parameter since n_subsamples != n_samples ({0} != {1}) while n_samples < n_features.)rB   rL   Ú
ValueErrorr9   r"   r!   r   Úrintr	   r   r]   )rb   rA   rM   rB   Ún_dimÚplus_1Úall_combinationsÚn_subpopulations           r+   Ú_check_subparamsz"TheilSenRegressor._check_subparamsb  sE  € ØÔ(ˆàÔð 	Ø ‘NˆEˆEàˆEàÐ#Ø˜iÒ'Ð'Ý ð-ß-3ªV°LÀ)Ñ-LÔ-Lñô ð ð ˜JÒ&Ð&Ø˜<Ò'Ð'Ø%)Ô%7Ð?˜T˜T¸R�FÝ$ðç!š6 &¨%°Ñ>Ô>ñô ð ð (ð   9Ò,Ð,Ý$ð(ç(.ª¨|¸YÑ(GÔ(Gñô ð ð -õ ˜u iÑ0Ô0ˆLå˜q¥"¤'­%°	¸<Ñ*HÔ*HÑ"IÔ"IÑJÔJÐÝ�c $Ô"8Ð:JÑKÔKÑLÔLˆà˜_Ð,Ð,r-   )Úprefer_skip_nested_validationc                 óÎ  ‡ ‡‡‡	‡
‡‡— t          ‰ j        ¦  «        Št          ‰ ‰‰d¬¦  «        \  ŠŠ‰j        \  Š
}‰                      ‰
|¦  «        \  Š‰ _        t          ‰
‰¦  «        ‰ _        ‰ j        r©t          d 
                    ‰ j        ¦  «        ¦  «         t          d 
                    ‰
¦  «        ¦  «         t          ‰ j        ‰
z  ¦  «        }t          d 
                    |¦  «        ¦  «         t          d 
                    ‰ j        ¦  «        ¦  «         t          j        t          ‰
‰¦  «        ¦  «        ‰ j        k    r+t!          t#          t%          ‰
¦  «        ‰¦  «        ¦  «        }n"ˆ
ˆˆfd„t%          ‰ j        ¦  «        D ¦   «         }t'          ‰ j        ¦  «        }t          j        ||¦  «        Š	 t-          |‰ j        ¬¦  «        ˆˆ	ˆ ˆfd	„t%          |¦  «        D ¦   «         ¦  «        }t          j        |¦  «        }t1          |‰ j        ‰ j        ¬
¦  «        \  ‰ _        }‰ j        r|d         ‰ _        |dd…         ‰ _        nd‰ _        |‰ _        ‰ S )aU  Fit linear model.

        Parameters
        ----------
        X : ndarray of shape (n_samples, n_features)
            Training data.
        y : ndarray of shape (n_samples,)
            Target values.

        Returns
        -------
        self : returns an instance of self.
            Fitted `TheilSenRegressor` estimator.
        T)Ú	y_numericzBreakdown point: {0}zNumber of samples: {0}zTolerable outliers: {0}zNumber of subpopulations: {0}c                 ó@   •— g | ]}‰                      ‰‰d ¬¦  «        ‘ŒS )F)ÚsizeÚreplace)Úchoice)Ú.0Ú_rA   rB   rZ   s     €€€r+   ú
<listcomp>z)TheilSenRegressor.fit.<locals>.<listcomp>ª  s>   ø€ ð ð ð àð ×#Ò# I°LÈ%Ð#ÑPÔPðð ð r-   )r^   r[   c              3   ón   •K  — | ]/} t          t          ¦  «        ‰‰‰|         ‰j        ¦  «        V — Œ0d S ra   )r   rT   rL   )rt   Újobr#   Ú
index_listrb   rJ   s     €€€€r+   ú	<genexpr>z(TheilSenRegressor.fit.<locals>.<genexpr>±  s\   øè è € ð @
ð @
àð �G•F‰OŒO˜A˜q *¨S¤/°4Ô3EÑFÔFð@
ð @
ð @
ð @
ð @
ð @
r-   )r2   r:   r   r   Nr   )r   rZ   r   r   rl   Ún_subpopulation_rC   Ú
breakdown_r[   Úprintr9   r   r   rg   r	   r]   Úlistr   r6   r   r^   Úarray_splitr   Úvstackr>   r2   r:   Ún_iter_rL   Ú
intercept_Úcoef_)rb   r#   rJ   rM   Útol_outliersrK   r^   rN   Úcoefsry   rA   rB   rZ   s   ```      @@@@r+   ÚfitzTheilSenRegressor.fit‡  sŠ  øøøøøøø€ õ  *¨$Ô*;Ñ<Ô<ˆÝ˜T 1 a°4Ð8Ñ8Ô8‰ˆˆ1Ø !¤Ñˆ	�:Ø.2×.CÒ.CØ�zñ/
ô /
Ñ+ˆ�dÔ+õ +¨9°lÑCÔCˆŒàŒ<ð 	QÝÐ(×/Ò/°´Ñ@Ô@ÑAÔAÐAÝÐ*×1Ò1°)Ñ<Ô<Ñ=Ô=Ð=Ý˜tœ°Ñ:Ñ;Ô;ˆLÝÐ+×2Ò2°<Ñ@Ô@ÑAÔAÐAÝÐ1×8Ò8¸Ô9NÑOÔOÑPÔPÐPõ Œ7•5˜ LÑ1Ô1Ñ2Ô2°dÔ6LÒLÐLÝ�<­¨iÑ(8Ô(8¸,ÑGÔGÑHÔHˆGˆGðð ð ð ð ð å˜tÔ4Ñ5Ô5ðñ ô ˆGõ
 " $¤+Ñ.Ô.ˆÝ”^ G¨VÑ4Ô4ˆ
Ø?•( &°$´,Ð?Ñ?Ô?ð @
ð @
ð @
ð @
ð @
ð @
ð @
å˜V‘}”}ð@
ñ @
ô @
ñ 
ô 
ˆõ ”)˜GÑ$Ô$ˆÝ-Ø˜dœm°´ð
ñ 
ô 
ÑˆŒ�eð Ôð 	Ø# AœhˆDŒOØ˜q˜r˜rœˆDŒJˆJà!ˆDŒOØˆDŒJàˆr-   )Ú__name__Ú
__module__Ú__qualname__Ú__doc__r   r   r   r_   ÚdictÚ__annotations__rc   rl   r   r†   r@   r-   r+   rV   rV   Ï   s  € € € € € € ðoð oðd $˜à&˜h t¨Q°¸VÐDÑDÔDÐEØ˜xÐ(Ø�X˜h¨¨4¸Ð?Ñ?Ô?Ð@Ø�˜˜s D°Ð8Ñ8Ô8Ð9Ø'Ð(Ø˜Ð"Ø�;ð
$ð 
$Ð˜Dð 
ð 
ñ 
ð ØØØØØØØðð ð ð ð ð*#-ð #-ð #-ðJ €\°Ð5Ñ5Ô5ð9ð 9ñ 6Ô5ð9ð 9ð 9r-   rV   )r.   r/   )*rŠ   r7   Ú	itertoolsr   Únumbersr   r   Únumpyr   Újoblibr   Úscipyr   Úscipy.linalg.lapackr   Úscipy.specialr	   Úsklearn.baser
   r   Úsklearn.exceptionsr   Úsklearn.linear_model._baser   Úsklearn.utilsr   Úsklearn.utils._param_validationr   Úsklearn.utils.parallelr   r   Úsklearn.utils.validationr   ÚfinfoÚdoubleÚepsr   r,   r>   rC   rT   rV   r@   r-   r+   ú<module>rž      s¼  ððð ð €€€Ø "Ð "Ð "Ð "Ð "Ð "Ø "Ð "Ð "Ð "Ð "Ð "Ð "Ð "à Ð Ð Ð Ø #Ð #Ð #Ð #Ð #Ð #Ø Ð Ð Ð Ð Ð Ø 0Ð 0Ð 0Ð 0Ð 0Ð 0Ø Ð Ð Ð Ð Ð à 5Ð 5Ð 5Ð 5Ð 5Ð 5Ð 5Ð 5Ø 1Ð 1Ð 1Ð 1Ð 1Ð 1Ø 2Ð 2Ð 2Ð 2Ð 2Ð 2Ø ,Ð ,Ð ,Ð ,Ð ,Ð ,Ø 4Ð 4Ð 4Ð 4Ð 4Ð 4Ø 4Ð 4Ð 4Ð 4Ð 4Ð 4Ð 4Ð 4Ø 2Ð 2Ð 2Ð 2Ð 2Ð 2àˆ2Œ8�B”IÑÔÔ"€ð0ð 0ð 0ðf5"ð 5"ð 5"ð 5"ðpð ð ð6)ð )ð )ðXrð rð rð rð r˜¨ñ rô rð rð rð rr-   