§
    rŠtj'3  ã                   óÞ   — d Z ddlZddlmZ ddlZddlmZ ddl	m
Z
mZmZ ddlmZ ddlmZmZ ddlmZ dd	lmZ dd
lmZ ddlmZmZ ddlmZ ddlmZ ddlm Z m!Z!  G d„ deee
¦  «        Z"dS )z!
Nearest Centroid Classification
é    N)ÚReal)Úsparse)ÚBaseEstimatorÚClassifierMixinÚ_fit_context)Ú#DiscriminantAnalysisPredictionMixin)Úpairwise_distancesÚpairwise_distances_argmin)ÚLabelEncoder)Úget_tags)Úavailable_if)ÚIntervalÚ
StrOptions)Úcheck_classification_targets)Úcsc_median_axis_0)Úcheck_is_fittedÚvalidate_datac                   ór  ‡ — e Zd ZU dZ eddh¦  «        g eeddd¬¦  «        dgd ed	d
h¦  «        gdœZee	d<   	 ddd
dœd„Z
 ed¬¦  «        d„ ¦   «         Zˆ fd„Zd„ Zd„ Z  ee¦  «        ej        ¦  «        Z  ee¦  «        ej        ¦  «        Z  ee¦  «        ej        ¦  «        Zˆ fd„Zˆ xZS )ÚNearestCentroidaž  Nearest centroid classifier.

    Each class is represented by its centroid, with test samples classified to
    the class with the nearest centroid.

    Read more in the :ref:`User Guide <nearest_centroid_classifier>`.

    Parameters
    ----------
    metric : {"euclidean", "manhattan"}, default="euclidean"
        Metric to use for distance computation.

        If `metric="euclidean"`, the centroid for the samples corresponding to each
        class is the arithmetic mean, which minimizes the sum of squared L1 distances.
        If `metric="manhattan"`, the centroid is the feature-wise median, which
        minimizes the sum of L1 distances.

        .. versionchanged:: 1.5
            All metrics but `"euclidean"` and `"manhattan"` were deprecated and
            now raise an error.

        .. versionchanged:: 0.19
            `metric='precomputed'` was deprecated and now raises an error

    shrink_threshold : float, default=None
        Threshold for shrinking centroids to remove features.

    priors : {"uniform", "empirical"} or array-like of shape (n_classes,),         default="uniform"
        The class prior probabilities. By default, the class proportions are
        inferred from the training data.

        .. versionadded:: 1.6

    Attributes
    ----------
    centroids_ : array-like of shape (n_classes, n_features)
        Centroid of each class.

    classes_ : array of shape (n_classes,)
        The unique classes labels.

    n_features_in_ : int
        Number of features seen during :term:`fit`.

        .. versionadded:: 0.24

    feature_names_in_ : ndarray of shape (`n_features_in_`,)
        Names of features seen during :term:`fit`. Defined only when `X`
        has feature names that are all strings.

        .. versionadded:: 1.0

    deviations_ : ndarray of shape (n_classes, n_features)
        Deviations (or shrinkages) of the centroids of each class from the
        overall centroid. Equal to eq. (18.4) if `shrink_threshold=None`,
        else (18.5) p. 653 of [2]. Can be used to identify features used
        for classification.

        .. versionadded:: 1.6

    within_class_std_dev_ : ndarray of shape (n_features,)
        Pooled or within-class standard deviation of input data.

        .. versionadded:: 1.6

    class_prior_ : ndarray of shape (n_classes,)
        The class prior probabilities.

        .. versionadded:: 1.6

    See Also
    --------
    KNeighborsClassifier : Nearest neighbors classifier.

    Notes
    -----
    When used for text classification with tf-idf vectors, this classifier is
    also known as the Rocchio classifier.

    References
    ----------
    [1] Tibshirani, R., Hastie, T., Narasimhan, B., & Chu, G. (2002). Diagnosis of
    multiple cancer types by shrunken centroids of gene expression. Proceedings
    of the National Academy of Sciences of the United States of America,
    99(10), 6567-6572. The National Academy of Sciences.

    [2] Hastie, T., Tibshirani, R., Friedman, J. (2009). The Elements of Statistical
    Learning Data Mining, Inference, and Prediction. 2nd Edition. New York, Springer.

    Examples
    --------
    >>> from sklearn.neighbors import NearestCentroid
    >>> import numpy as np
    >>> X = np.array([[-1, -1], [-2, -1], [-3, -2], [1, 1], [2, 1], [3, 2]])
    >>> y = np.array([1, 1, 1, 2, 2, 2])
    >>> clf = NearestCentroid()
    >>> clf.fit(X, y)
    NearestCentroid()
    >>> print(clf.predict([[-0.8, -1]]))
    [1]
    Ú	manhattanÚ	euclideanr   NÚneither)Úclosedz
array-likeÚ	empiricalÚuniform©ÚmetricÚshrink_thresholdÚpriorsÚ_parameter_constraints)r   r   c                ó0   — || _         || _        || _        d S )Nr   )Úselfr   r   r   s       úa/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/sklearn/neighbors/_nearest_centroid.pyÚ__init__zNearestCentroid.__init__Š   s   € ð ˆŒØ 0ˆÔØˆŒˆˆó    T)Úprefer_skip_nested_validationc                 ó"  — | j         dk    rt          | ||dg¬¦  «        \  }}n6t          | ¦  «        j        j        rdnd}t          | |||ddg¬¦  «        \  }}t          j        |¦  «        }t          |¦  «         |j        \  }}t          ¦   «         }| 
                    |¦  «        }|j        x| _        }	|	j        }
|
dk     rt          d	|
z  ¦  «        ‚| j        d
k    rPt          j        |d¬¦  «        \  }}t          j        |¦  «        t%          t'          |¦  «        ¦  «        z  | _        nJ| j        dk    r!t          j        d|
z  g|
z  ¦  «        | _        nt          j        | j        ¦  «        | _        | j        dk                          ¦   «         rt          d¦  «        ‚t          j        | j                             ¦   «         d¦  «        s@t3          j        dt6          ¦  «         | j        | j                             ¦   «         z  | _        t          j        |
|ft          j        ¬¦  «        | _        t          j        |
¦  «        }tA          |
¦  «        D ]¯}||k    }t          j        |¦  «        ||<   |rt          j!        |¦  «        d         }| j         dk    rE|s%t          j"        ||         d¬¦  «        | j        |<   ŒmtG          ||         ¦  «        | j        |<   Œ‹||          $                    d¬¦  «        | j        |<   Œ°t          j%        || j        |         z
  d¬¦  «        dz  }t          j%        t          j&        |                     d¬¦  «        ||
z
  z  ¦  «        d¬¦  «        | _'        t-          | j'        dk    ¦  «        rt3          j        d¦  «         d}|rdt          j(        | )                    d¬¦  «        | *                    d¬¦  «        z
   +                    ¦   «         dk    ¦  «        rt          |¦  «        ‚|s;t          j(        t          j,        |d¬¦  «        dk    ¦  «        rt          |¦  «        ‚| $                    d¬¦  «        }t          j&        d|z  d|z  z
  ¦  «        }| j'        t          j"        | j'        ¦  «        z   }| -                    t'          |¦  «        d¦  «        }||z  }t          j%        | j        |z
  |z  d¬¦  «        | _.        | j/        r™t          j0        | j.        ¦  «        }t          j1        | j.        ¦  «        | j/        z
  | _.        t          j2        | j.        dd| j.        ¬¦  «         | xj.        |z  c_.        || j.        z  }t          j%        ||z   d¬¦  «        | _        | S )a0  
        Fit the NearestCentroid model according to the given training data.

        Parameters
        ----------
        X : {array-like, sparse matrix} of shape (n_samples, n_features)
            Training vector, where `n_samples` is the number of samples and
            `n_features` is the number of features.
            Note that centroid shrinking cannot be used with sparse matrices.
        y : array-like of shape (n_samples,)
            Target values.

        Returns
        -------
        self : object
            Fitted estimator.
        r   Úcsc)Úaccept_sparseú	allow-nanTÚcsr)Úensure_all_finiter)   é   z>The number of classes has to be greater than one; got %d classr   )Úreturn_inverser   é   r   zpriors must be non-negativeg      ð?zAThe priors do not sum to 1. Normalizing such that it sums to one.©Údtype)ÚaxisF)Úcopyz†self.within_class_std_dev_ has at least 1 zero standard deviation.Inputs within the same classes for at least 1 feature are identical.z2All features have zero variance. Division by zero.N)Úout)3r   r   r   Ú
input_tagsÚ	allow_nanÚspÚissparser   Úshaper   Úfit_transformÚclasses_ÚsizeÚ
ValueErrorr   ÚnpÚuniqueÚbincountÚfloatÚlenÚclass_prior_ÚasarrayÚanyÚiscloseÚsumÚwarningsÚwarnÚUserWarningÚemptyÚfloat64Ú
centroids_ÚzerosÚrangeÚwhereÚmedianr   ÚmeanÚarrayÚsqrtÚwithin_class_std_dev_ÚallÚmaxÚminÚtoarrayÚptpÚreshapeÚdeviations_r   ÚsignÚabsÚclip)r"   ÚXÚyr,   Úis_X_sparseÚ	n_samplesÚ
n_featuresÚleÚy_indÚclassesÚ	n_classesÚ_Úclass_countsÚnkÚ	cur_classÚcenter_maskÚvarianceÚerr_msgÚdataset_centroid_ÚmÚsÚmmÚmsÚsignsÚmsds                            r#   ÚfitzNearestCentroid.fit•   su  € ð* Œ;˜+Ò%Ð%Ý   q¨!¸E¸7ÐCÑCÔC‰DˆAˆqˆqõ  (¨™~œ~Ô8ÔBÐL��Èð õ !ØØØØ"3Ø$ e˜nðñ ô ‰DˆAˆqõ ”k !‘n”nˆÝ$ QÑ'Ô'Ð'à !¤Ñˆ	�:Ý‰^Œ^ˆØ× Ò  Ñ#Ô#ˆØ"$¤+Ð-ˆŒ˜Ø”Lˆ	Ø�qŠ=ˆ=ÝØPØññô ð ð
 Œ;˜+Ò%Ð%Ý œi¨¸$Ð?Ñ?Ô?‰OˆAˆ|Ý "¤¨LÑ 9Ô 9½EÅ#ÀaÁ&Ä&¹M¼MÑ IˆDÔÐØŒ[˜IÒ%Ð%Ý "¤
¨A°	©M¨?¸YÑ+FÑ GÔ GˆDÔÐå "¤
¨4¬;Ñ 7Ô 7ˆDÔàÔ Ò!×&Ò&Ñ(Ô(ð 	<ÝÐ:Ñ;Ô;Ð;ÝŒz˜$Ô+×/Ò/Ñ1Ô1°3Ñ7Ô7ð 	LÝŒMØSÝñô ð ð !%Ô 1°DÔ4E×4IÒ4IÑ4KÔ4KÑ KˆDÔõ œ( I¨zÐ#:Å"Ä*ÐMÑMÔMˆŒõ ŒX�iÑ Ô ˆå˜yÑ)Ô)ð 	Ið 	IˆIØ 9Ò,ˆKÝœF ;Ñ/Ô/ˆBˆy‰MØð 7Ý œh {Ñ3Ô3°AÔ6�àŒ{˜kÒ)Ð)à"ð SÝ13´¸1¸[¼>ÐPQÐ1RÑ1RÔ1R�D”O IÑ.Ð.å1BÀ1À[Ä>Ñ1RÔ1R�D”O IÑ.Ð.à-.¨{¬^×-@Ò-@ÀaÐ-@Ñ-HÔ-H�” 	Ñ*Ð*õ ”8˜A ¤°Ô 6Ñ6¸UÐCÑCÔCÀqÑHˆÝ%'¤XÝŒG�H—L’L a�LÑ(Ô(¨I¸	Ñ,AÑBÑCÔCÈ%ð&
ñ &
ô &
ˆÔ"õ ˆtÔ)¨QÒ.Ñ/Ô/ð 	ÝŒMðWñô ð ð
 GˆØð 	&�2œ6 1§5¢5¨a 5¡=¤=°1·5²5¸a°5±=´=Ñ#@×"IÒ"IÑ"KÔ"KÈqÒ"PÑQÔQð 	&Ý˜WÑ%Ô%Ð%Øð 	&¥¤­¬¨q°qÐ(9Ñ(9Ô(9¸QÒ(>Ñ!?Ô!?ð 	&Ý˜WÑ%Ô%Ð%àŸFšF¨˜F™NœNÐåŒG�S˜2‘X #¨	¡/Ñ2Ñ3Ô3ˆð Ô&­¬°4Ô3MÑ)NÔ)NÑNˆØ�YŠY•s˜1‘v”v˜qÑ!Ô!ˆØ�!‰VˆÝœ8ØŒ_Ð0Ñ0°BÑ6¸Uð
ñ 
ô 
ˆÔð
 Ô ð 	LÝ”G˜DÔ,Ñ-Ô-ˆEÝ!œv dÔ&6Ñ7Ô7¸$Ô:OÑOˆDÔÝŒG�DÔ$ a¨°4Ô3CÐDÑDÔDÐDØÐÔ Ñ%ÐÔà�tÔ'Ñ'ˆCÝ œhÐ'8¸3Ñ'>ÀUÐKÑKÔKˆDŒOØˆr%   c                 ó–  •— t          | ¦  «         t          j        | j        dt	          | j        ¦  «        z  ¦  «                             ¦   «         rXt          | ¦  «        j        j	        rdnd}t          | ||dd¬¦  «        }| j        t          || j        | j        ¬¦  «                 S t          ¦   «                              |¦  «        S )a€  Perform classification on an array of test vectors `X`.

        The predicted class `C` for each sample in `X` is returned.

        Parameters
        ----------
        X : {array-like, sparse matrix} of shape (n_samples, n_features)
            Input data.

        Returns
        -------
        y_pred : ndarray of shape (n_samples,)
            The predicted classes.
        r/   r*   Tr+   F)r,   r)   Úreset©r   )r   r>   rF   rC   rB   r;   rV   r   r5   r6   r   r
   rM   r   ÚsuperÚpredict)r"   r`   r,   Ú	__class__s      €r#   r|   zNearestCentroid.predict  sÈ   ø€ õ 	˜ÑÔÐÝŒ:�dÔ'¨­S°´Ñ-?Ô-?Ñ)?Ñ@Ô@×DÒDÑFÔFð 	&õ  (¨™~œ~Ô8ÔBÐL��Èð õ ØØØ"3Ø#Øðñ ô ˆAð ”=Ý)¨!¨T¬_ÀTÄ[ÐQÑQÔQôð õ ‘7”7—?’? 1Ñ%Ô%Ð%r%   c           	      ó¶  — t          | d¦  «         t          | |dddt          j        ¬¦  «        }t          j        |j        d         | j        j        ft          j        ¬¦  «        }| j        dk    }|d d …|fxx         | j        |         z  cc<   | j	         
                    ¦   «         }|d d …|fxx         | j        |         z  cc<   t          | j        j        ¦  «        D ]v}t          |||g         | j        ¬¦  «                             ¦   «         }|d	z  }t          j        | d
t          j        | j        |         ¦  «        z  z   ¦  «        |d d …|f<   Œw|S )NrM   TFr+   )r3   ry   r)   r1   r   r0   rz   r-   g       @)r   r   r>   rL   rK   r9   r;   r<   rU   rM   r3   rO   r	   r   ÚravelÚsqueezeÚlogrC   )r"   r`   ÚX_normalizedÚdiscriminant_scoreÚmaskÚcentroids_normalizedÚ	class_idxÚ	distancess           r#   Ú_decision_functionz"NearestCentroid._decision_function5  s…  € å˜˜lÑ+Ô+Ð+å$Ø�!˜$ e¸5ÍÌ
ð
ñ 
ô 
ˆõ  œXØÔ Ô" D¤MÔ$6Ð7½r¼zð
ñ 
ô 
Ðð Ô)¨QÒ.ˆØ�Q�Q�Q˜�WÐÐÔ Ô!;¸DÔ!AÑAÐÐÑØ#œ×3Ò3Ñ5Ô5ÐØ˜Q˜Q˜Q ˜WÐ%Ð%Ô%¨Ô)CÀDÔ)IÑIÐ%Ð%Ñ%å˜tœ}Ô1Ñ2Ô2ð 	ð 	ˆIÝ*ØÐ2°I°;Ô?ÈÌðñ ô çŠe‰gŒgð ð ˜!‰OˆIÝ/1¬zØ�
˜S¥2¤6¨$Ô*;¸IÔ*FÑ#GÔ#GÑGÑGñ0ô 0Ð˜q˜q˜q )˜|Ñ,Ð,ð "Ð!r%   c                 ó   — | j         dk    S )Nr   rz   )r"   s    r#   Ú_check_euclidean_metricz'NearestCentroid._check_euclidean_metricQ  s   € ØŒ{˜kÒ)Ð)r%   c                 óŠ   •— t          ¦   «                              ¦   «         }| j        dk    |j        _        d|j        _        |S )NÚnan_euclideanT)r{   Ú__sklearn_tags__r   r5   r6   r   )r"   Útagsr}   s     €r#   r�   z NearestCentroid.__sklearn_tags__`  s8   ø€ Ý‰wŒw×'Ò'Ñ)Ô)ˆØ$(¤K°?Ò$BˆŒÔ!Ø!%ˆŒÔØˆr%   )r   )Ú__name__Ú
__module__Ú__qualname__Ú__doc__r   r   r   r    ÚdictÚ__annotations__r$   r   rw   r|   rˆ   rŠ   r   r   Údecision_functionÚpredict_probaÚpredict_log_probar�   Ú__classcell__)r}   s   @r#   r   r      s¬  ø€ € € € € € ðeð eðP �:˜{¨KÐ8Ñ9Ô9Ð:Ø%˜X d¨A¨t¸IÐFÑFÔFÈÐMØ  ¨[¸)Ð,DÑ!EÔ!EÐFð$ð $Ð˜Dð ð ñ ð ð	ð Øð	ð 	ð 	ð 	ð 	ð €\°Ð5Ñ5Ô5ð{ð {ñ 6Ô5ð{ðz &ð  &ð  &ð  &ð  &ðD"ð "ð "ð8*ð *ð *ð >˜˜Ð%<Ñ=Ô=Ø+Ô=ñô Ðð :�L�LÐ!8Ñ9Ô9Ø+Ô9ñô €Mð >˜˜Ð%<Ñ=Ô=Ø+Ô=ñô Ððð ð ð ð ð ð ð ð r%   r   )#r’   rH   Únumbersr   Únumpyr>   Úscipyr   r7   Úsklearn.baser   r   r   Úsklearn.discriminant_analysisr   Úsklearn.metrics.pairwiser	   r
   Úsklearn.preprocessingr   Úsklearn.utilsr   Úsklearn.utils._available_ifr   Úsklearn.utils._param_validationr   r   Úsklearn.utils.multiclassr   Úsklearn.utils.sparsefuncsr   Úsklearn.utils.validationr   r   r   © r%   r#   ú<module>r§      s_  ððð ð €€€Ø Ð Ð Ð Ð Ð à Ð Ð Ð Ø Ð Ð Ð Ð Ð à EÐ EÐ EÐ EÐ EÐ EÐ EÐ EÐ EÐ EØ MÐ MÐ MÐ MÐ MÐ MØ RÐ RÐ RÐ RÐ RÐ RÐ RÐ RØ .Ð .Ð .Ð .Ð .Ð .Ø "Ð "Ð "Ð "Ð "Ð "Ø 4Ð 4Ð 4Ð 4Ð 4Ð 4Ø @Ð @Ð @Ð @Ð @Ð @Ð @Ð @Ø AÐ AÐ AÐ AÐ AÐ AØ 7Ð 7Ð 7Ð 7Ð 7Ð 7Ø CÐ CÐ CÐ CÐ CÐ CÐ CÐ CðJð Jð Jð Jð JØ'¨¸-ñJô Jð Jð Jð Jr%   