§
    qŠtjÃæ  ã                   ó   — d Z ddlZddlZddlmZ ddlmZmZm	Z	m
Z
mZmZmZmZmZmZmZ ddlmZmZmZmZmZmZ ddlmZ ddlmZ ddlmZmZm Z  dd	l!m"Z" dd
l#m$Z$  G d„ d¦  «        Z% G d„ de%¦  «        Z& G d„ de%¦  «        Z' G d„ de%¦  «        Z( G d„ de%¦  «        Z) G d„ de%¦  «        Z* G d„ de%¦  «        Z+ G d„ de%¦  «        Z, G d„ de%¦  «        Z- G d„ de%¦  «        Z. G d„ d e%¦  «        Z/ G d!„ d"e%¦  «        Z0e&e'e(e)e*e+e,e.e/e0d#œ
Z1 G d$„ d%¦  «        Z2d&„ Z3 G d'„ d(e2e.¦  «        Z4 G d)„ d*e2e/¦  «        Z5 G d+„ d,e2e*¦  «        Z6dS )-z¹
This module contains loss classes suitable for fitting.

It is not part of the public API.
Specific losses are used for regression, binary classification or multiclass
classification.
é    N©Úxlogy)ÚCyAbsoluteErrorÚCyExponentialLossÚCyHalfBinomialLossÚCyHalfGammaLossÚCyHalfMultinomialLossÚCyHalfPoissonLossÚCyHalfSquaredErrorÚCyHalfTweedieLossÚCyHalfTweedieLossIdentityÚCyHuberLossÚCyPinballLoss)ÚHalfLogitLinkÚIdentityLinkÚIntervalÚ	LogitLinkÚLogLinkÚMultinomialLogit)Úone_hot)Úcheck_scalar)Ú_averageÚ
_logsumexpÚ_ravel)Úsoftmax)Ú_weighted_percentilec                   ó˜   — e Zd ZdZdZdZdd„Zd„ Zd„ Z	 	 	 dd	„Z		 	 	 	 dd
„Z
	 	 	 dd„Z	 	 	 	 dd„Zdd„Zdd„Zdd„Zej        dfd„ZdS )ÚBaseLossaÒ  Base class for a loss function of 1-dimensional targets.

    Conventions:

        - y_true.shape = sample_weight.shape = (n_samples,)
        - y_pred.shape = raw_prediction.shape = (n_samples,)
        - If is_multiclass is true (multiclass classification), then
          y_pred.shape = raw_prediction.shape = (n_samples, n_classes)
          Note that this corresponds to the return value of decision_function.

    y_true, y_pred, sample_weight and raw_prediction must either be all float64
    or all float32.
    gradient and hessian must be either both float64 or both float32.

    Note that y_pred = link.inverse(raw_prediction).

    Specific loss classes can inherit specific link classes to satisfy
    BaseLink's abstractmethods.

    Parameters
    ----------
    closs: CyLossFunction
        For example, a CyLossFunction; hence the name "c"loss.
    link : BaseLink
    sample_weight : {None, ndarray}
        If sample_weight is None, the hessian might be constant.
    n_classes : {None, int}
        The number of classes for classification, else None.
    xp : module, default=None
        Array namespace module.
    device : device, default=None
        A device object (see the "Device Support" section of the array API spec).

    Attributes
    ----------
    closs: CyLossFunction
        For example, a CyLossFunction; hence the name "c"loss.
    link : BaseLink
    n_classes : {None, int}
        The number of classes for classification, else None.
    xp : module or None
        Array namespace module. Ignored by the Cython implementation.
    device : device or None
        A device object. Ignored by the Cython implementation.
    interval_y_true : Interval
        Valid interval for y_true
    interval_y_pred : Interval
        Valid Interval for y_pred
    differentiable : bool
        Indicates whether or not loss function is differentiable in
        raw_prediction everywhere.
    approx_hessian : bool
        Indicates whether the hessian is approximated or exact. If,
        approximated, it should be larger or equal to the exact one.
    constant_hessian : bool
        Indicates whether the hessian is one for this loss.
    is_multiclass : bool
        Indicates whether n_classes > 2 is allowed.
    TFNc                 óâ   — || _         || _        || _        || _        || _        d| _        d| _        t          t          j	         t          j	        dd¦  «        | _
        | j        j        | _        d S )NF)ÚclossÚlinkÚ	n_classesÚxpÚdeviceÚapprox_hessianÚconstant_hessianr   ÚnpÚinfÚinterval_y_trueÚinterval_y_pred)Úselfr    r!   r"   r#   r$   s         úP/var/www/html/CA-Chatbot/venv/lib/python3.11/site-packages/sklearn/_loss/loss.pyÚ__init__zBaseLoss.__init__–   sd   € ØˆŒ
ØˆŒ	Ø"ˆŒØˆŒØˆŒØ#ˆÔØ %ˆÔÝ'­¬¨µ´¸ÀÑFÔFˆÔØ#œyÔ8ˆÔÐÐó    c                 ó6   — | j                              |¦  «        S ©zuReturn True if y is in the valid range of y_true.

        Parameters
        ----------
        y : ndarray
        )r)   Úincludes©r+   Úys     r,   Úin_y_true_rangezBaseLoss.in_y_true_range¡   ó   € ð Ô#×,Ò,¨QÑ/Ô/Ð/r.   c                 ó6   — | j                              |¦  «        S )zuReturn True if y is in the valid range of y_pred.

        Parameters
        ----------
        y : ndarray
        )r*   r1   r2   s     r,   Úin_y_pred_rangezBaseLoss.in_y_pred_rangeª   r5   r.   é   c                 óÒ   — |€t          j        |¦  «        }|j        dk    r&|j        d         dk    r|                     d¦  «        }| j                             |||||¬¦  «         |S )aJ  Compute the pointwise loss value for each input.

        Parameters
        ----------
        y_true : C-contiguous array of shape (n_samples,)
            Observed, true target values.
        raw_prediction : C-contiguous array of shape (n_samples,) or array of             shape (n_samples, n_classes)
            Raw prediction values (in link space).
        sample_weight : None or C-contiguous array of shape (n_samples,)
            Sample weights.
        loss_out : None or C-contiguous array of shape (n_samples,)
            A location into which the result is stored. If None, a new array
            might be created.
        n_threads : int, default=1
            Might use openmp thread parallelism.

        Returns
        -------
        loss : array of shape (n_samples,)
            Element-wise loss function.
        Né   r8   ©Úy_trueÚraw_predictionÚsample_weightÚloss_outÚ	n_threads)r'   Ú
empty_likeÚndimÚshapeÚsqueezer    Úloss©r+   r<   r=   r>   r?   r@   s         r,   rE   zBaseLoss.loss³   s   € ð< ÐÝ”} VÑ,Ô,ˆHàÔ !Ò#Ð#¨Ô(<¸QÔ(?À1Ò(DÐ(DØ+×3Ò3°AÑ6Ô6ˆNàŒ
�ŠØØ)Ø'ØØð 	ñ 	
ô 	
ð 	
ð ˆr.   c                 óÚ  — |€G|€)t          j        |¦  «        }t          j        |¦  «        }n9t          j        ||j        ¬¦  «        }n|€t          j        ||j        ¬¦  «        }|j        dk    r&|j        d         dk    r|                     d¦  «        }|j        dk    r&|j        d         dk    r|                     d¦  «        }| j                             ||||||¬¦  «         ||fS )a¯  Compute loss and gradient w.r.t. raw_prediction for each input.

        Parameters
        ----------
        y_true : C-contiguous array of shape (n_samples,)
            Observed, true target values.
        raw_prediction : C-contiguous array of shape (n_samples,) or array of             shape (n_samples, n_classes)
            Raw prediction values (in link space).
        sample_weight : None or C-contiguous array of shape (n_samples,)
            Sample weights.
        loss_out : None or C-contiguous array of shape (n_samples,)
            A location into which the loss is stored. If None, a new array
            might be created.
        gradient_out : None or C-contiguous array of shape (n_samples,) or array             of shape (n_samples, n_classes)
            A location into which the gradient is stored. If None, a new array
            might be created.
        n_threads : int, default=1
            Might use openmp thread parallelism.

        Returns
        -------
        loss : array of shape (n_samples,)
            Element-wise loss function.

        gradient : array of shape (n_samples,) or (n_samples, n_classes)
            Element-wise gradients.
        N©Údtyper:   r8   )r<   r=   r>   r?   Úgradient_outr@   )r'   rA   rI   rB   rC   rD   r    Úloss_gradient)r+   r<   r=   r>   r?   rJ   r@   s          r,   rK   zBaseLoss.loss_gradientà   s  € ðL ÐØÐ#Ýœ=¨Ñ0Ô0�Ý!œ}¨^Ñ<Ô<��åœ=¨°|Ô7IÐJÑJÔJ��ØÐ!Ýœ=¨¸x¼~ÐNÑNÔNˆLð Ô !Ò#Ð#¨Ô(<¸QÔ(?À1Ò(DÐ(DØ+×3Ò3°AÑ6Ô6ˆNØÔ Ò!Ð! lÔ&8¸Ô&;¸qÒ&@Ð&@Ø'×/Ò/°Ñ2Ô2ˆLàŒ
× Ò ØØ)Ø'ØØ%Øð 	!ñ 	
ô 	
ð 	
ð ˜Ð%Ð%r.   c                 ó4  — |€t          j        |¦  «        }|j        dk    r&|j        d         dk    r|                     d¦  «        }|j        dk    r&|j        d         dk    r|                     d¦  «        }| j                             |||||¬¦  «         |S )aª  Compute gradient of loss w.r.t raw_prediction for each input.

        Parameters
        ----------
        y_true : C-contiguous array of shape (n_samples,)
            Observed, true target values.
        raw_prediction : C-contiguous array of shape (n_samples,) or array of             shape (n_samples, n_classes)
            Raw prediction values (in link space).
        sample_weight : None or C-contiguous array of shape (n_samples,)
            Sample weights.
        gradient_out : None or C-contiguous array of shape (n_samples,) or array             of shape (n_samples, n_classes)
            A location into which the result is stored. If None, a new array
            might be created.
        n_threads : int, default=1
            Might use openmp thread parallelism.

        Returns
        -------
        gradient : array of shape (n_samples,) or (n_samples, n_classes)
            Element-wise gradients.
        Nr:   r8   )r<   r=   r>   rJ   r@   )r'   rA   rB   rC   rD   r    Úgradient©r+   r<   r=   r>   rJ   r@   s         r,   rM   zBaseLoss.gradient  s·   € ð> ÐÝœ=¨Ñ8Ô8ˆLð Ô !Ò#Ð#¨Ô(<¸QÔ(?À1Ò(DÐ(DØ+×3Ò3°AÑ6Ô6ˆNØÔ Ò!Ð! lÔ&8¸Ô&;¸qÒ&@Ð&@Ø'×/Ò/°Ñ2Ô2ˆLàŒ
×ÒØØ)Ø'Ø%Øð 	ñ 	
ô 	
ð 	
ð Ðr.   c                 ó   — |€@|€)t          j        |¦  «        }t          j        |¦  «        }n+t          j        |¦  «        }n|€t          j        |¦  «        }|j        dk    r&|j        d         dk    r|                     d¦  «        }|j        dk    r&|j        d         dk    r|                     d¦  «        }|j        dk    r&|j        d         dk    r|                     d¦  «        }| j                             ||||||¬¦  «         ||fS )aÿ  Compute gradient and hessian of loss w.r.t raw_prediction.

        Parameters
        ----------
        y_true : C-contiguous array of shape (n_samples,)
            Observed, true target values.
        raw_prediction : C-contiguous array of shape (n_samples,) or array of             shape (n_samples, n_classes)
            Raw prediction values (in link space).
        sample_weight : None or C-contiguous array of shape (n_samples,)
            Sample weights.
        gradient_out : None or C-contiguous array of shape (n_samples,) or array             of shape (n_samples, n_classes)
            A location into which the gradient is stored. If None, a new array
            might be created.
        hessian_out : None or C-contiguous array of shape (n_samples,) or array             of shape (n_samples, n_classes)
            A location into which the hessian is stored. If None, a new array
            might be created.
        n_threads : int, default=1
            Might use openmp thread parallelism.

        Returns
        -------
        gradient : arrays of shape (n_samples,) or (n_samples, n_classes)
            Element-wise gradients.

        hessian : arrays of shape (n_samples,) or (n_samples, n_classes)
            Element-wise hessians.
        Nr:   r8   )r<   r=   r>   rJ   Úhessian_outr@   )r'   rA   rB   rC   rD   r    Úgradient_hessian)r+   r<   r=   r>   rJ   rP   r@   s          r,   rQ   zBaseLoss.gradient_hessianP  s0  € ðN ÐØÐ"Ý!œ}¨^Ñ<Ô<�Ý œm¨NÑ;Ô;��å!œ}¨[Ñ9Ô9��ØÐ Ýœ-¨Ñ5Ô5ˆKð Ô !Ò#Ð#¨Ô(<¸QÔ(?À1Ò(DÐ(DØ+×3Ò3°AÑ6Ô6ˆNØÔ Ò!Ð! lÔ&8¸Ô&;¸qÒ&@Ð&@Ø'×/Ò/°Ñ2Ô2ˆLØÔ˜qÒ Ð  [Ô%6°qÔ%9¸QÒ%>Ð%>Ø%×-Ò-¨aÑ0Ô0ˆKàŒ
×#Ò#ØØ)Ø'Ø%Ø#Øð 	$ñ 	
ô 	
ð 	
ð ˜[Ð(Ð(r.   c           	      ó^   — t          j        |                      ||dd|¬¦  «        |¬¦  «        S )a{  Compute the weighted average loss.

        Parameters
        ----------
        y_true : C-contiguous array of shape (n_samples,)
            Observed, true target values.
        raw_prediction : C-contiguous array of shape (n_samples,) or array of             shape (n_samples, n_classes)
            Raw prediction values (in link space).
        sample_weight : None or C-contiguous array of shape (n_samples,)
            Sample weights.
        n_threads : int, default=1
            Might use openmp thread parallelism.

        Returns
        -------
        loss : float
            Mean or averaged loss function.
        Nr;   ©Úweights)r'   ÚaveragerE   )r+   r<   r=   r>   r@   s        r,   Ú__call__zBaseLoss.__call__’  sG   € õ( ŒzØ�IŠIØØ-Ø"ØØ#ð ñ ô ð "ð	
ñ 	
ô 	
ð 		
r.   c                 ó   — t          j        ||d¬¦  «        }dt          j        |j        ¦  «        j        z  }| j        j        t           j         k    rd}n(| j        j        r| j        j        }n| j        j        |z   }| j        j	        t           j        k    rd}n(| j        j
        r| j        j	        }n| j        j	        |z
  }|€|€| j                             |¦  «        S | j                             t          j        |||¦  «        ¦  «        S )a#  Compute raw_prediction of an intercept-only model.

        This can be used as initial estimates of predictions, i.e. before the
        first iteration in fit.

        Parameters
        ----------
        y_true : array-like of shape (n_samples,)
            Observed, true target values.
        sample_weight : None or array of shape (n_samples,)
            Sample weights.

        Returns
        -------
        raw_prediction : numpy scalar or array of shape (n_classes,)
            Raw predictions of an intercept-only model.
        r   ©rT   Úaxisé
   N)r'   rU   ÚfinforI   Úepsr*   Úlowr(   Úlow_inclusiveÚhighÚhigh_inclusiver!   Úclip)r+   r<   r>   Úy_predr\   Úa_minÚa_maxs          r,   Úfit_intercept_onlyzBaseLoss.fit_intercept_only±  sþ   € õ( ”˜F¨MÀÐBÑBÔBˆØ•2”8˜FœLÑ)Ô)Ô-Ñ-ˆàÔÔ#­¬ wÒ.Ð.ØˆEˆEØÔ!Ô/ð 	3ØÔ(Ô,ˆEˆEàÔ(Ô,¨sÑ2ˆEàÔÔ$­¬Ò.Ð.ØˆEˆEØÔ!Ô0ð 	4ØÔ(Ô-ˆEˆEàÔ(Ô-°Ñ3ˆEàˆ=˜U˜]Ø”9—>’> &Ñ)Ô)Ð)à”9—>’>¥"¤'¨&°%¸Ñ"?Ô"?Ñ@Ô@Ð@r.   c                 ó*   — t          j        |¦  «        S )a(  Calculate term dropped in loss.

        With this term added, the loss of perfect predictions is zero.

        Parameters
        ----------
        y_true : array-like of shape (n_samples,)
            Observed, true target values.

        sample_weight : None or array of shape (n_samples,), default=None
            Sample weights.

        Returns
        -------
        constant : ndarray of shape (n_samples,)
            Constant value to be added to raw predictions so that the loss
            of perfect predictions becomes zero.
        )r'   Ú
zeros_like©r+   r<   r>   s      r,   Úconstant_to_optimal_zeroz!BaseLoss.constant_to_optimal_zeroÛ  s   € õ& Œ}˜VÑ$Ô$Ð$r.   ÚFc                 ó$  — |t           j        t           j        fvrt          d|› d�¦  «        ‚| j        r
|| j        f}n|f}t          j        |||¬¦  «        }| j        rt          j        d|¬¦  «        }nt          j        |||¬¦  «        }||fS )au  Initialize arrays for gradients and hessians.

        Unless hessians are constant, arrays are initialized with undefined values.

        Parameters
        ----------
        n_samples : int
            The number of samples, usually passed to `fit()`.
        dtype : {np.float64, np.float32}, default=np.float64
            The dtype of the arrays gradient and hessian.
        order : {'C', 'F'}, default='F'
            Order of the arrays gradient and hessian. The default 'F' makes the arrays
            contiguous along samples.

        Returns
        -------
        gradient : C-contiguous array of shape (n_samples,) or array of shape             (n_samples, n_classes)
            Empty array (allocated but not initialized) to be used as argument
            gradient_out.
        hessian : C-contiguous array of shape (n_samples,), array of shape
            (n_samples, n_classes) or shape (1,)
            Empty (allocated but not initialized) array to be used as argument
            hessian_out.
            If constant_hessian is True (e.g. `HalfSquaredError`), the array is
            initialized to ``1``.
        zCValid options for 'dtype' are np.float32 and np.float64. Got dtype=z	 instead.)rC   rI   Úorder)r8   )rC   rI   )	r'   Úfloat32Úfloat64Ú
ValueErrorÚis_multiclassr"   Úemptyr&   Úones)r+   Ú	n_samplesrI   rl   rC   rM   Úhessians          r,   Úinit_gradient_and_hessianz"BaseLoss.init_gradient_and_hessianð  s¾   € ð8 �œ¥R¤ZÐ0Ð0Ð0Ýð.Ø"ð.ð .ð .ñô ð ð
 Ôð 	!Ø ¤Ð/ˆEˆEà�LˆEÝ”8 %¨u¸EÐBÑBÔBˆàÔ ð 	Fõ
 ”g D°Ð6Ñ6Ô6ˆGˆGå”h U°%¸uÐEÑEÔEˆGà˜Ð Ð r.   ©NNN©NNr8   ©NNNr8   ©Nr8   ©N)Ú__name__Ú
__module__Ú__qualname__Ú__doc__Údifferentiablerp   r-   r4   r7   rE   rK   rM   rQ   rV   re   ri   r'   rn   ru   © r.   r,   r   r   M   sE  € € € € € ð:ð :ðJ €NØ€Mð	9ð 	9ð 	9ð 	9ð0ð 0ð 0ð0ð 0ð 0ð ØØð+ð +ð +ð +ðb ØØØð=&ð =&ð =&ð =&ðF ØØð/ð /ð /ð /ðj ØØØð@)ð @)ð @)ð @)ðD
ð 
ð 
ð 
ð>(Að (Að (Að (AðT%ð %ð %ð %ð* :<¼È3ð 1!ð 1!ð 1!ð 1!ð 1!ð 1!r.   r   c                   ó$   ‡ — e Zd ZdZdˆ fd„	Zˆ xZS )ÚHalfSquaredErroraÜ  Half squared error with identity link, for regression.

    Domain:
    y_true and y_pred all real numbers

    Link:
    y_pred = raw_prediction

    For a given sample x_i, half squared error is defined as::

        loss(x_i) = 0.5 * (y_true_i - raw_prediction_i)**2

    The factor of 0.5 simplifies the computation of gradients and results in a
    unit hessian (and is consistent with what is done in LightGBM). It is also
    half the Normal distribution deviance.
    Nc                 ó”   •— t          ¦   «                              t          ¦   «         t          ¦   «         ||¬¦  «         |d u | _        d S )N©r    r!   r#   r$   )Úsuperr-   r   r   r&   ©r+   r>   r#   r$   Ú	__class__s       €r,   r-   zHalfSquaredError.__init__6  sL   ø€ Ý‰Œ×ÒÝ$Ñ&Ô&­\©^¬^ÀÈ6ð 	ñ 	
ô 	
ð 	
ð !.°Ð 5ˆÔÐÐr.   rv   ©r{   r|   r}   r~   r-   Ú__classcell__©r‡   s   @r,   r‚   r‚   $  sG   ø€ € € € € ðð ð"6ð 6ð 6ð 6ð 6ð 6ð 6ð 6ð 6ð 6r.   r‚   c                   ó0   ‡ — e Zd ZdZdZdˆ fd„	Zdd„Zˆ xZS )ÚAbsoluteErroraÖ  Absolute error with identity link, for regression.

    Domain:
    y_true and y_pred all real numbers

    Link:
    y_pred = raw_prediction

    For a given sample x_i, the absolute error is defined as::

        loss(x_i) = |y_true_i - raw_prediction_i|

    Note that the exact hessian = 0 almost everywhere (except at one point, therefore
    differentiable = False). Optimization routines like in HGBT, however, need a
    hessian > 0. Therefore, we assign 1.
    FNc                 ó¢   •— t          ¦   «                              t          ¦   «         t          ¦   «         ||¬¦  «         d| _        |d u | _        d S )Nr„   T)r…   r-   r   r   r%   r&   r†   s       €r,   r-   zAbsoluteError.__init__Q  sT   ø€ Ý‰Œ×ÒÝ!Ñ#Ô#­,©.¬.¸RÈð 	ñ 	
ô 	
ð 	
ð #ˆÔØ -°Ð 5ˆÔÐÐr.   c                 óT   — |€t          j        |d¬¦  «        S t          ||d¦  «        S )ú•Compute raw_prediction of an intercept-only model.

        This is the weighted median of the target, i.e. over the samples
        axis=0.
        Nr   ©rY   é2   )r'   Úmedianr   rh   s      r,   re   z AbsoluteError.fit_intercept_onlyX  s1   € ð Ð Ý”9˜V¨!Ð,Ñ,Ô,Ð,å'¨°¸rÑBÔBÐBr.   rv   rz   ©r{   r|   r}   r~   r   r-   re   r‰   rŠ   s   @r,   rŒ   rŒ   =  sj   ø€ € € € € ðð ð" €Nð6ð 6ð 6ð 6ð 6ð 6ð	Cð 	Cð 	Cð 	Cð 	Cð 	Cð 	Cð 	Cr.   rŒ   c                   ó0   ‡ — e Zd ZdZdZdˆ fd„	Zdd„Zˆ xZS )	ÚPinballLossa  Quantile loss aka pinball loss, for regression.

    Domain:
    y_true and y_pred all real numbers
    quantile in (0, 1)

    Link:
    y_pred = raw_prediction

    For a given sample x_i, the pinball loss is defined as::

        loss(x_i) = rho_{quantile}(y_true_i - raw_prediction_i)

        rho_{quantile}(u) = u * (quantile - 1_{u<0})
                          = -u *(1 - quantile)  if u < 0
                             u * quantile       if u >= 0

    Note: 2 * PinballLoss(quantile=0.5) equals AbsoluteError().

    Note that the exact hessian = 0 almost everywhere (except at one point, therefore
    differentiable = False). Optimization routines like in HGBT, however, need a
    hessian > 0. Therefore, we assign 1.

    Additional Attributes
    ---------------------
    quantile : float
        The quantile level of the quantile to be estimated. Must be in range (0, 1).
    FNç      à?c                 óþ   •— t          |dt          j        ddd¬¦  «         t          ¦   «                              t          t          |¦  «        ¬¦  «        t          ¦   «         ||¬¦  «         d| _        |d u | _	        d S )	NÚquantiler   r8   Úneither©Útarget_typeÚmin_valÚmax_valÚinclude_boundaries)r˜   r„   T)
r   ÚnumbersÚRealr…   r-   r   Úfloatr   r%   r&   )r+   r>   r˜   r#   r$   r‡   s        €r,   r-   zPinballLoss.__init__„  s�   ø€ ÝØØÝœØØØ(ð	
ñ 	
ô 	
ð 	
õ 	‰Œ×ÒÝ­¨x©¬Ð9Ñ9Ô9Ý‘”ØØð	 	ñ 	
ô 	
ð 	
ð #ˆÔØ -°Ð 5ˆÔÐÐr.   c                 óŠ   — |€$t          j        |d| j        j        z  d¬¦  «        S t	          ||d| j        j        z  ¦  «        S )r�   Néd   r   r�   )r'   Ú
percentiler    r˜   r   rh   s      r,   re   zPinballLoss.fit_intercept_only–  sN   € ð Ð Ý”= ¨¨t¬zÔ/BÑ)BÈÐKÑKÔKÐKå'Ø˜ s¨T¬ZÔ-@Ñ'@ñô ð r.   )Nr–   NNrz   r“   rŠ   s   @r,   r•   r•   d  sb   ø€ € € € € ðð ð: €Nð6ð 6ð 6ð 6ð 6ð 6ð$ð ð ð ð ð ð ð r.   r•   c                   ó2   ‡ — e Zd ZdZdZ	 dˆ fd„	Zd	d„Zˆ xZS )
Ú	HuberLossaê  Huber loss, for regression.

    Domain:
    y_true and y_pred all real numbers
    quantile in (0, 1)

    Link:
    y_pred = raw_prediction

    For a given sample x_i, the Huber loss is defined as::

        loss(x_i) = 1/2 * abserr**2            if abserr <= delta
                    delta * (abserr - delta/2) if abserr > delta

        abserr = |y_true_i - raw_prediction_i|
        delta = quantile(abserr, self.quantile)

    Note: HuberLoss(quantile=1) equals HalfSquaredError and HuberLoss(quantile=0)
    equals delta * (AbsoluteError() - delta/2).

    Additional Attributes
    ---------------------
    quantile : float
        The quantile level which defines the breaking point `delta` to distinguish
        between absolute error and squared error. Must be in range (0, 1).

     Reference
    ---------
    .. [1] Friedman, J.H. (2001). :doi:`Greedy function approximation: A gradient
      boosting machine <10.1214/aos/1013203451>`.
      Annals of Statistics, 29, 1189-1232.
    FNçÍÌÌÌÌÌì?r–   c                 ó  •— t          |dt          j        ddd¬¦  «         || _        t	          ¦   «                              t          t          |¦  «        ¬¦  «        t          ¦   «         ||¬¦  «         d| _	        d	| _
        d S )
Nr˜   r   r8   r™   rš   )Údeltar„   TF)r   rŸ   r    r˜   r…   r-   r   r¡   r   r%   r&   )r+   r>   r˜   r©   r#   r$   r‡   s         €r,   r-   zHuberLoss.__init__È  s“   ø€ õ 	ØØÝœØØØ(ð	
ñ 	
ô 	
ð 	
ð !ˆŒÝ‰Œ×ÒÝ¥E¨%¡L¤LÐ1Ñ1Ô1Ý‘”ØØð	 	ñ 	
ô 	
ð 	
ð #ˆÔØ %ˆÔÐÐr.   c                 ó   — |€t          j        |dd¬¦  «        }nt          ||d¦  «        }||z
  }t          j        |¦  «        t          j        | j        j        t          j        |¦  «        ¦  «        z  }|t          j        ||¬¦  «        z   S )r�   Nr‘   r   r�   rS   )	r'   r¤   r   ÚsignÚminimumr    r©   ÚabsrU   )r+   r<   r>   r’   ÚdiffÚterms         r,   re   zHuberLoss.fit_intercept_onlyÝ  s�   € ð Ð Ý”] 6¨2°AÐ6Ñ6Ô6ˆFˆFå)¨&°-ÀÑDÔDˆFØ˜‰ˆÝŒw�t‰}Œ}�rœz¨$¬*Ô*:½B¼FÀ4¹L¼LÑIÔIÑIˆØ�œ
 4°Ð?Ñ?Ô?Ñ?Ð?r.   )Nr§   r–   NNrz   r“   rŠ   s   @r,   r¦   r¦   ¤  sr   ø€ € € € € ðð ðB €Nð LPð&ð &ð &ð &ð &ð &ð*@ð @ð @ð @ð @ð @ð @ð @r.   r¦   c                   ó,   ‡ — e Zd ZdZdˆ fd„	Zdd„Zˆ xZS )ÚHalfPoissonLossaƒ  Half Poisson deviance loss with log-link, for regression.

    Domain:
    y_true in non-negative real numbers
    y_pred in positive real numbers

    Link:
    y_pred = exp(raw_prediction)

    For a given sample x_i, half the Poisson deviance is defined as::

        loss(x_i) = y_true_i * log(y_true_i/exp(raw_prediction_i))
                    - y_true_i + exp(raw_prediction_i)

    Half the Poisson deviance is actually the negative log-likelihood up to
    constant terms (not involving raw_prediction) and simplifies the
    computation of the gradients.
    We also skip the constant term `y_true_i * log(y_true_i) - y_true_i`.
    Nc                 óÄ   •— t          ¦   «                              t          ¦   «         t          ¦   «         ||¬¦  «         t	          dt
          j        dd¦  «        | _        d S )Nr„   r   TF)r…   r-   r
   r   r   r'   r(   r)   r†   s       €r,   r-   zHalfPoissonLoss.__init__  sW   ø€ Ý‰Œ×ÒÝ#Ñ%Ô%­G©I¬I¸"ÀVð 	ñ 	
ô 	
ð 	
õ  (¨­2¬6°4¸Ñ?Ô?ˆÔÐÐr.   c                 ó:   — t          ||¦  «        |z
  }|�||z  }|S rz   r   ©r+   r<   r>   r¯   s       r,   ri   z(HalfPoissonLoss.constant_to_optimal_zero  s+   € Ý�V˜VÑ$Ô$ vÑ-ˆØÐ$Ø�MÑ!ˆDØˆr.   rv   rz   ©r{   r|   r}   r~   r-   ri   r‰   rŠ   s   @r,   r±   r±   ð  sa   ø€ € € € € ðð ð(@ð @ð @ð @ð @ð @ðð ð ð ð ð ð ð r.   r±   c                   ó,   ‡ — e Zd ZdZdˆ fd„	Zdd„Zˆ xZS )ÚHalfGammaLossaV  Half Gamma deviance loss with log-link, for regression.

    Domain:
    y_true and y_pred in positive real numbers

    Link:
    y_pred = exp(raw_prediction)

    For a given sample x_i, half Gamma deviance loss is defined as::

        loss(x_i) = log(exp(raw_prediction_i)/y_true_i)
                    + y_true/exp(raw_prediction_i) - 1

    Half the Gamma deviance is actually proportional to the negative log-
    likelihood up to constant terms (not involving raw_prediction) and
    simplifies the computation of the gradients.
    We also skip the constant term `-log(y_true_i) - 1`.
    Nc                 óÄ   •— t          ¦   «                              t          ¦   «         t          ¦   «         ||¬¦  «         t	          dt
          j        dd¦  «        | _        d S )Nr„   r   F)r…   r-   r   r   r   r'   r(   r)   r†   s       €r,   r-   zHalfGammaLoss.__init__&  sM   ø€ Ý‰Œ×Ò�Ñ0Ô0µw±y´yÀRÐPVÐÑWÔWÐWÝ'¨­2¬6°5¸%Ñ@Ô@ˆÔÐÐr.   c                 óD   — t          j        |¦  «         dz
  }|�||z  }|S ry   )r'   Úlogr´   s       r,   ri   z&HalfGammaLoss.constant_to_optimal_zero*  s+   € Ý”�v‘”ˆ Ñ"ˆØÐ$Ø�MÑ!ˆDØˆr.   rv   rz   rµ   rŠ   s   @r,   r·   r·     sa   ø€ € € € € ðð ð&Að Að Að Að Að Aðð ð ð ð ð ð ð r.   r·   c                   ó,   ‡ — e Zd ZdZdˆ fd„	Zdd„Zˆ xZS )ÚHalfTweedieLossaÿ  Half Tweedie deviance loss with log-link, for regression.

    Domain:
    y_true in real numbers for power <= 0
    y_true in non-negative real numbers for 0 < power < 2
    y_true in positive real numbers for 2 <= power
    y_pred in positive real numbers
    power in real numbers

    Link:
    y_pred = exp(raw_prediction)

    For a given sample x_i, half Tweedie deviance loss with p=power is defined
    as::

        loss(x_i) = max(y_true_i, 0)**(2-p) / (1-p) / (2-p)
                    - y_true_i * exp(raw_prediction_i)**(1-p) / (1-p)
                    + exp(raw_prediction_i)**(2-p) / (2-p)

    Taking the limits for p=0, 1, 2 gives HalfSquaredError with a log link,
    HalfPoissonLoss and HalfGammaLoss.

    We also skip constant terms, but those are different for p=0, 1, 2.
    Therefore, the loss is not continuous in `power`.

    Note furthermore that although no Tweedie distribution exists for
    0 < power < 1, it still gives a strictly consistent scoring function for
    the expectation.
    Nç      ø?c                 óÄ  •— t          ¦   «                              t          t          |¦  «        ¬¦  «        t	          ¦   «         ||¬¦  «         | j        j        dk    r.t          t          j	         t          j	        dd¦  «        | _
        d S | j        j        dk     r#t          dt          j	        dd¦  «        | _
        d S t          dt          j	        dd¦  «        | _
        d S ©N)Úpowerr„   r   Fr:   T)r…   r-   r   r¡   r   r    rÀ   r   r'   r(   r)   ©r+   r>   rÀ   r#   r$   r‡   s        €r,   r-   zHalfTweedieLoss.__init__P  sÅ   ø€ Ý‰Œ×ÒÝ#­%°©,¬,Ð7Ñ7Ô7Ý‘”ØØð	 	ñ 	
ô 	
ð 	
ð Œ:Ô˜qÒ Ð Ý#+­R¬V¨GµR´V¸UÀEÑ#JÔ#JˆDÔ Ð Ð ØŒZÔ Ò!Ð!Ý#+¨A­r¬v°t¸UÑ#CÔ#CˆDÔ Ð Ð å#+¨A­r¬v°u¸eÑ#DÔ#DˆDÔ Ð Ð r.   c                 óÌ  — | j         j        dk    r#t          ¦   «                              ||¬¦  «        S | j         j        dk    r#t	          ¦   «                              ||¬¦  «        S | j         j        dk    r#t          ¦   «                              ||¬¦  «        S | j         j        }t          j        t          j        |d¦  «        d|z
  ¦  «        d|z
  z  d|z
  z  }|�||z  }|S )Nr   )r<   r>   r8   r:   )r    rÀ   r‚   ri   r±   r·   r'   Úmaximum)r+   r<   r>   Úpr¯   s        r,   ri   z(HalfTweedieLoss.constant_to_optimal_zero^  s   € ØŒ:Ô˜qÒ Ð Ý#Ñ%Ô%×>Ò>Ø¨]ð ?ñ ô ð ð ŒZÔ Ò"Ð"Ý"Ñ$Ô$×=Ò=Ø¨]ð >ñ ô ð ð ŒZÔ Ò"Ð"Ý ‘?”?×;Ò;Ø¨]ð <ñ ô ð ð ”
Ô ˆAÝ”8�BœJ v¨qÑ1Ô1°1°q±5Ñ9Ô9¸QÀ¹UÑCÀqÈ1ÁuÑMˆDØÐ(Ø˜Ñ%�ØˆKr.   ©Nr½   NNrz   rµ   rŠ   s   @r,   r¼   r¼   1  sa   ø€ € € € € ðð ð<Eð Eð Eð Eð Eð Eðð ð ð ð ð ð ð r.   r¼   c                   ó$   ‡ — e Zd ZdZdˆ fd„	Zˆ xZS )ÚHalfTweedieLossIdentityan  Half Tweedie deviance loss with identity link, for regression.

    Domain:
    y_true in real numbers for power <= 0
    y_true in non-negative real numbers for 0 < power < 2
    y_true in positive real numbers for 2 <= power
    y_pred in positive real numbers for power != 0
    y_pred in real numbers for power = 0
    power in real numbers

    Link:
    y_pred = raw_prediction

    For a given sample x_i, half Tweedie deviance loss with p=power is defined
    as::

        loss(x_i) = max(y_true_i, 0)**(2-p) / (1-p) / (2-p)
                    - y_true_i * raw_prediction_i**(1-p) / (1-p)
                    + raw_prediction_i**(2-p) / (2-p)

    Note that the minimum value of this loss is 0.

    Note furthermore that although no Tweedie distribution exists for
    0 < power < 1, it still gives a strictly consistent scoring function for
    the expectation.
    Nr½   c                 ó~  •— t          ¦   «                              t          t          |¦  «        ¬¦  «        t	          ¦   «         ||¬¦  «         | j        j        dk    r-t          t          j	         t          j	        dd¦  «        | _
        nS| j        j        dk     r"t          dt          j	        dd¦  «        | _
        n!t          dt          j	        dd¦  «        | _
        | j        j        dk    r.t          t          j	         t          j	        dd¦  «        | _        d S t          dt          j	        dd¦  «        | _        d S r¿   )r…   r-   r   r¡   r   r    rÀ   r   r'   r(   r)   r*   rÁ   s        €r,   r-   z HalfTweedieLossIdentity.__init__�  s	  ø€ Ý‰Œ×ÒÝ+µ%¸±,´,Ð?Ñ?Ô?Ý‘”ØØð	 	ñ 	
ô 	
ð 	
ð Œ:Ô˜qÒ Ð Ý#+­R¬V¨GµR´V¸UÀEÑ#JÔ#JˆDÔ Ð ØŒZÔ Ò!Ð!Ý#+¨A­r¬v°t¸UÑ#CÔ#CˆDÔ Ð å#+¨A­r¬v°u¸eÑ#DÔ#DˆDÔ àŒ:Ô˜qÒ Ð Ý#+­R¬V¨GµR´V¸UÀEÑ#JÔ#JˆDÔ Ð Ð å#+¨A­r¬v°u¸eÑ#DÔ#DˆDÔ Ð Ð r.   rÅ   rˆ   rŠ   s   @r,   rÇ   rÇ   s  sQ   ø€ € € € € ðð ð6Eð Eð Eð Eð Eð Eð Eð Eð Eð Er.   rÇ   c                   ó2   ‡ — e Zd ZdZdˆ fd„	Zdd„Zd„ Zˆ xZS )ÚHalfBinomialLossaY  Half Binomial deviance loss with logit link, for binary classification.

    This is also know as binary cross entropy, log-loss and logistic loss.

    Domain:
    y_true in [0, 1], i.e. regression on the unit interval
    y_pred in (0, 1), i.e. boundaries excluded

    Link:
    y_pred = expit(raw_prediction)

    For a given sample x_i, half Binomial deviance is defined as the negative
    log-likelihood of the Binomial/Bernoulli distribution and can be expressed
    as::

        loss(x_i) = log(1 + exp(raw_pred_i)) - y_true_i * raw_pred_i

    See The Elements of Statistical Learning, by Hastie, Tibshirani, Friedman,
    section 4.4.1 (about logistic regression).

    Note that the formulation works for classification, y = {0, 1}, as well as
    logistic regression, y = [0, 1].
    If you add `constant_to_optimal_zero` to the loss, you get half the
    Bernoulli/binomial deviance.

    More details: Inserting the predicted probability y_pred = expit(raw_prediction)
    in the loss gives the well known::

        loss(x_i) = - y_true_i * log(y_pred_i) - (1 - y_true_i) * log(1 - y_pred_i)
    Nc                 ó²   •— t          ¦   «                              t          ¦   «         t          ¦   «         d||¬¦  «         t	          dddd¦  «        | _        d S ©Nr:   ©r    r!   r"   r#   r$   r   r8   T)r…   r-   r   r   r   r)   r†   s       €r,   r-   zHalfBinomialLoss.__init__Ã  s[   ø€ Ý‰Œ×ÒÝ$Ñ&Ô&Ý‘”ØØØð 	ñ 	
ô 	
ð 	
õ  (¨¨1¨d°DÑ9Ô9ˆÔÐÐr.   c                 ób   — t          ||¦  «        t          d|z
  d|z
  ¦  «        z   }|�||z  }|S ry   r   r´   s       r,   ri   z)HalfBinomialLoss.constant_to_optimal_zeroÍ  s=   € å�V˜VÑ$Ô$¥u¨Q°©Z¸¸V¹Ñ'DÔ'DÑDˆØÐ$Ø�MÑ!ˆDØˆr.   c                 ó&  — |j         dk    r&|j        d         dk    r|                     d¦  «        }t          j        |j        d         df|j        ¬¦  «        }| j                             |¦  «        |dd…df<   d|dd…df         z
  |dd…df<   |S ©a=  Predict probabilities.

        Parameters
        ----------
        raw_prediction : array of shape (n_samples,) or (n_samples, 1)
            Raw prediction values (in link space).

        Returns
        -------
        proba : array of shape (n_samples, 2)
            Element-wise class probabilities.
        r:   r8   r   rH   N©rB   rC   rD   r'   rq   rI   r!   Úinverse©r+   r=   Úprobas      r,   Úpredict_probazHalfBinomialLoss.predict_probaÔ  ó¢   € ð Ô !Ò#Ð#¨Ô(<¸QÔ(?À1Ò(DÐ(DØ+×3Ò3°AÑ6Ô6ˆNÝ”˜.Ô.¨qÔ1°1Ð5¸^Ô=QÐRÑRÔRˆØ”i×'Ò'¨Ñ7Ô7ˆˆaˆaˆa�ˆd‰Ø˜%    1 œ+‘oˆˆaˆaˆa�ˆd‰Øˆr.   rv   rz   ©r{   r|   r}   r~   r-   ri   rÕ   r‰   rŠ   s   @r,   rÊ   rÊ   £  sj   ø€ € € € € ðð ð>:ð :ð :ð :ð :ð :ðð ð ð ðð ð ð ð ð ð r.   rÊ   c                   óL   ‡ — e Zd ZdZdZdˆ fd„	Zd„ Zdd„Zd„ Z	 	 	 	 dd
„Z	ˆ xZ
S )ÚHalfMultinomialLossa/  Categorical cross-entropy loss, for multiclass classification.

    Domain:
    y_true in {0, 1, 2, 3, .., n_classes - 1}
    y_pred has n_classes elements, each element in (0, 1)

    Link:
    y_pred = softmax(raw_prediction)

    Note: We assume y_true to be already label encoded. The inverse link is
    softmax. But the full link function is the symmetric multinomial logit
    function.

    For a given sample x_i, the categorical cross-entropy loss is defined as
    the negative log-likelihood of the multinomial distribution, it
    generalizes the binary cross-entropy to more than 2 classes::

        loss_i = log(sum(exp(raw_pred_{i, k}), k=0..n_classes-1))
                - sum(y_true_{i, k} * raw_pred_{i, k}, k=0..n_classes-1)

    See [1].

    Note that for the hessian, we calculate only the diagonal part in the
    classes: If the full hessian for classes k and l and sample i is H_i_k_l,
    we calculate H_i_k_k, i.e. k=l.

    Parameters
    ----------
    sample_weight : {None, ndarray}
        If sample_weight is None, the hessian might be constant.

    n_classes : {None, int}
        The number of classes for classification, else None.

    xp : module or None
        Array namespace module. Ignored by the Cython implementation.

    device : device or None
        A device object. Ignored by the Cython implementation.

    References
    ----------
    .. [1] :arxiv:`Simon, Noah, J. Friedman and T. Hastie.
        "A Blockwise Descent Algorithm for Group-penalized Multiresponse and
        Multinomial Regression".
        <1311.6529>`
    TNé   c                 ó  •— t          ¦   «                              t          ¦   «         t          ¦   «         |||¬¦  «         t	          dt
          j        dd¦  «        | _        t	          dddd¦  «        | _        d | _	        d | _
        d | _        d S )NrÍ   r   TFr8   )r…   r-   r	   r   r   r'   r(   r)   r*   Úclass_indexing_offsetsÚ
y_true_intÚy_true_one_hot©r+   r>   r"   r#   r$   r‡   s        €r,   r-   zHalfMultinomialLoss.__init__  sŽ   ø€ Ý‰Œ×ÒÝ'Ñ)Ô)Ý!Ñ#Ô#ØØØð 	ñ 	
ô 	
ð 	
õ  (¨­2¬6°4¸Ñ?Ô?ˆÔÝ'¨¨1¨e°UÑ;Ô;ˆÔð '+ˆÔ#ØˆŒØ"ˆÔÐÐr.   c                 ó–   — | j                              |¦  «        o/t          j        |                     t
          ¦  «        |k    ¦  «        S r0   )r)   r1   r'   ÚallÚastypeÚintr2   s     r,   r4   z#HalfMultinomialLoss.in_y_true_range.  s9   € ð Ô#×,Ò,¨QÑ/Ô/ÐNµB´F¸1¿8º8ÅC¹=¼=ÈAÒ;MÑ4NÔ4NÐNr.   c                 óš  — t          j        | j        |j        ¬¦  «        }t          j        |j        ¦  «        j        }t          | j        ¦  «        D ]B}t          j        ||k    |d¬¦  «        ||<   t          j        ||         |d|z
  ¦  «        ||<   ŒC| j	         	                    |ddd…f         ¦  «         
                    d¦  «        S )a-  Compute raw_prediction of an intercept-only model.

        This is the softmax of the weighted average of the target, i.e. over
        the samples axis=0.

        Parameters
        ----------
        y_true : array-like of shape (n_samples,)
            Observed, true target values.

        sample_weight : None or array of shape (n_samples,), default=None
            Sample weights.

        Returns
        -------
        raw_prediction : numpy scalar or array of shape (n_classes,)
            Raw predictions of an intercept-only model.
        rH   r   rX   r8   Néÿÿÿÿ)r'   Úzerosr"   rI   r[   r\   ÚrangerU   ra   r!   Úreshape)r+   r<   r>   Úoutr\   Úks         r,   re   z&HalfMultinomialLoss.fit_intercept_only7  s¶   € õ& Œh�t”~¨V¬\Ð:Ñ:Ô:ˆÝŒh�v”|Ñ$Ô$Ô(ˆÝ�t”~Ñ&Ô&ð 	3ð 	3ˆAÝ”Z ¨!¢°]ÈÐKÑKÔKˆC�‰FÝ”W˜S œV S¨!¨c©'Ñ2Ô2ˆC�‰FˆFØŒy�~Š~˜c $¨¨¨ 'œlÑ+Ô+×3Ò3°BÑ7Ô7Ð7r.   c                 ó6   — | j                              |¦  «        S )a=  Predict probabilities.

        Parameters
        ----------
        raw_prediction : array of shape (n_samples, n_classes)
            Raw prediction values (in link space).

        Returns
        -------
        proba : array of shape (n_samples, n_classes)
            Element-wise class probabilities.
        )r!   rÒ   )r+   r=   s     r,   rÕ   z!HalfMultinomialLoss.predict_probaQ  s   € ð Œy× Ò  Ñ0Ô0Ð0r.   r8   c                 óú   — |€@|€)t          j        |¦  «        }t          j        |¦  «        }n+t          j        |¦  «        }n|€t          j        |¦  «        }| j                             ||||||¬¦  «         ||fS )aK  Compute gradient and class probabilities fow raw_prediction.

        Parameters
        ----------
        y_true : C-contiguous array of shape (n_samples,)
            Observed, true target values.
        raw_prediction : array of shape (n_samples, n_classes)
            Raw prediction values (in link space).
        sample_weight : None or C-contiguous array of shape (n_samples,)
            Sample weights.
        gradient_out : None or array of shape (n_samples, n_classes)
            A location into which the gradient is stored. If None, a new array
            might be created.
        proba_out : None or array of shape (n_samples, n_classes)
            A location into which the class probabilities are stored. If None,
            a new array might be created.
        n_threads : int, default=1
            Might use openmp thread parallelism.

        Returns
        -------
        gradient : array of shape (n_samples, n_classes)
            Element-wise gradients.

        proba : array of shape (n_samples, n_classes)
            Element-wise class probabilities.
        N)r<   r=   r>   rJ   Ú	proba_outr@   )r'   rA   r    Úgradient_proba)r+   r<   r=   r>   rJ   rí   r@   s          r,   rî   z"HalfMultinomialLoss.gradient_proba`  s•   € ðH ÐØÐ Ý!œ}¨^Ñ<Ô<�ÝœM¨.Ñ9Ô9�	�	å!œ}¨YÑ7Ô7��ØÐÝœ lÑ3Ô3ˆIàŒ
×!Ò!ØØ)Ø'Ø%ØØð 	"ñ 	
ô 	
ð 	
ð ˜YÐ&Ð&r.   ©NrÚ   NNrz   rx   )r{   r|   r}   r~   rp   r-   r4   re   rÕ   rî   r‰   rŠ   s   @r,   rÙ   rÙ   ê  s¦   ø€ € € € € ð.ð .ð` €Mð#ð #ð #ð #ð #ð #ð"Oð Oð Oð8ð 8ð 8ð 8ð41ð 1ð 1ð& ØØØð5'ð 5'ð 5'ð 5'ð 5'ð 5'ð 5'ð 5'r.   rÙ   c                   ó2   ‡ — e Zd ZdZdˆ fd„	Zdd„Zd„ Zˆ xZS )ÚExponentialLossa"  Exponential loss with (half) logit link, for binary classification.

    This is also know as boosting loss.

    Domain:
    y_true in [0, 1], i.e. regression on the unit interval
    y_pred in (0, 1), i.e. boundaries excluded

    Link:
    y_pred = expit(2 * raw_prediction)

    For a given sample x_i, the exponential loss is defined as::

        loss(x_i) = y_true_i * exp(-raw_pred_i)) + (1 - y_true_i) * exp(raw_pred_i)

    See:
    - J. Friedman, T. Hastie, R. Tibshirani.
      "Additive logistic regression: a statistical view of boosting (With discussion
      and a rejoinder by the authors)." Ann. Statist. 28 (2) 337 - 407, April 2000.
      https://doi.org/10.1214/aos/1016218223
    - A. Buja, W. Stuetzle, Y. Shen. (2005).
      "Loss Functions for Binary Class Probability Estimation and Classification:
      Structure and Applications."

    Note that the formulation works for classification, y = {0, 1}, as well as
    "exponential logistic" regression, y = [0, 1].
    Note that this is a proper scoring rule, but without it's canonical link.

    More details: Inserting the predicted probability
    y_pred = expit(2 * raw_prediction) in the loss gives::

        loss(x_i) = y_true_i * sqrt((1 - y_pred_i) / y_pred_i)
            + (1 - y_true_i) * sqrt(y_pred_i / (1 - y_pred_i))
    Nc                 ó²   •— t          ¦   «                              t          ¦   «         t          ¦   «         d||¬¦  «         t	          dddd¦  «        | _        d S rÌ   )r…   r-   r   r   r   r)   r†   s       €r,   r-   zExponentialLoss.__init__¼  s[   ø€ Ý‰Œ×ÒÝ#Ñ%Ô%Ý‘”ØØØð 	ñ 	
ô 	
ð 	
õ  (¨¨1¨d°DÑ9Ô9ˆÔÐÐr.   c                 óN   — dt          j        |d|z
  z  ¦  «        z  }|�||z  }|S )Néþÿÿÿr8   )r'   Úsqrtr´   s       r,   ri   z(ExponentialLoss.constant_to_optimal_zeroÆ  s4   € à•B”G˜F a¨&¡jÑ1Ñ2Ô2Ñ2ˆØÐ$Ø�MÑ!ˆDØˆr.   c                 ó&  — |j         dk    r&|j        d         dk    r|                     d¦  «        }t          j        |j        d         df|j        ¬¦  «        }| j                             |¦  «        |dd…df<   d|dd…df         z
  |dd…df<   |S rÐ   rÑ   rÓ   s      r,   rÕ   zExponentialLoss.predict_probaÍ  rÖ   r.   rv   rz   r×   rŠ   s   @r,   rñ   rñ   ˜  sk   ø€ € € € € ð!ð !ðF:ð :ð :ð :ð :ð :ðð ð ð ðð ð ð ð ð ð r.   rñ   )
Úsquared_errorÚabsolute_errorÚpinball_lossÚ
huber_lossÚpoisson_lossÚ
gamma_lossÚtweedie_lossÚbinomial_lossÚmultinomial_lossÚexponential_lossc                   óJ   — e Zd ZdZ	 	 dd„Z	 	 	 d	d„Z	 	 	 	 d
d„Z	 	 	 d	d„ZdS )ÚArrayAPILossMixina¤  Mixin for loss classes that are compatible with the array API.

    Currently this mixin redefines methods:
    - __call__(...)
    - loss(...)
    - loss_gradient(...)
    - gradient(...)

    such that they work according to the array API specification.
    It uses the attributes self.xp and self.device from BaseLoss and it assumes that
    methods self._compute_loss and self._compute_gradient are implemented.
    Nr8   c                 óz   — |                       ||d¬¦  «        }t          t          ||| j        ¬¦  «        ¦  «        S )a˜  Compute the weighted average loss for the array API losses.

        Parameters
        ----------
        y_true : C-contiguous array of shape (n_samples,)
            Observed, true target values.
        raw_prediction : C-contiguous array of shape (n_samples,) or array of             shape (n_samples, n_classes)
            Raw prediction values (in link space).
        sample_weight : None or C-contiguous array of shape (n_samples,)
            Sample weights.
        n_threads : int, default=1
            Ignored by the array API implementation.

        Returns
        -------
        loss : float
            Mean or averaged loss function.
        N©r<   r=   r>   )rT   r#   )rE   r¡   r   r#   )r+   r<   r=   r>   r@   Úloss_xps         r,   rV   zArrayAPILossMixin.__call__ÿ  sD   € ð4 —)’)Ø¨.Èð ñ 
ô 
ˆõ •X˜g¨}ÀÄÐIÑIÔIÑJÔJÐJr.   c                 ó2   — |                       |||¬¦  «        S )a  Compute the pointwise loss value for each input.

        Parameters
        ----------
        y_true : C-contiguous array of shape (n_samples,)
            Observed, true target values.
        raw_prediction : C-contiguous array of shape (n_samples,) or array of             shape (n_samples, n_classes)
            Raw prediction values (in link space).
        sample_weight : None or C-contiguous array of shape (n_samples,)
            Sample weights.
        loss_out : None or C-contiguous array of shape (n_samples,)
            Ignored by the array API implementation.
        n_threads : int, default=1
            Ignored by the array API implementation.

        Returns
        -------
        loss : array of shape (n_samples,)
            Element-wise loss function.
        r  )Ú_compute_lossrF   s         r,   rE   zArrayAPILossMixin.loss  s*   € ð: ×!Ò!ØØ)Ø'ð "ñ 
ô 
ð 	
r.   c                 ój   — |                       |||¬¦  «        }|                      |||¬¦  «        }||fS )aG  Compute loss and gradient w.r.t. raw_prediction for each input.

        Parameters
        ----------
        y_true : C-contiguous array of shape (n_samples,)
            Observed, true target values.
        raw_prediction : C-contiguous array of shape (n_samples,) or array of             shape (n_samples, n_classes)
            Raw prediction values (in link space).
        sample_weight : None or C-contiguous array of shape (n_samples,)
            Sample weights.
        loss_out : None or C-contiguous array of shape (n_samples,)
            Ignored by the array API implementation.
        gradient_out : None or C-contiguous array of shape (n_samples,) or array             of shape (n_samples, n_classes)
            Ignored by the array API implementation.
        n_threads : int, default=1
            Ignored by the array API implementation.

        Returns
        -------
        loss : array of shape (n_samples,)
            Element-wise loss function.

        gradient : array of shape (n_samples,) or (n_samples, n_classes)
            Element-wise gradients.
        r  )r  Ú_compute_gradient)	r+   r<   r=   r>   r?   rJ   r@   rE   rM   s	            r,   rK   zArrayAPILossMixin.loss_gradientA  sY   € ðH ×!Ò!ØØ)Ø'ð "ñ 
ô 
ˆð
 ×)Ò)ØØ)Ø'ð *ñ 
ô 
ˆð
 �Xˆ~Ðr.   c                 ó2   — |                       |||¬¦  «        S )ax  Compute gradient of loss w.r.t raw_prediction for each input.

        Parameters
        ----------
        y_true : C-contiguous array of shape (n_samples,)
            Observed, true target values.
        raw_prediction : C-contiguous array of shape (n_samples,) or array of             shape (n_samples, n_classes)
            Raw prediction values (in link space).
        sample_weight : None or C-contiguous array of shape (n_samples,)
            Sample weights.
        gradient_out : None or C-contiguous array of shape (n_samples,) or array             of shape (n_samples, n_classes)
            Ignored by the array API implementation.
        n_threads : int, default=1
            Ignored by the array API implementation.

        Returns
        -------
        gradient : array of shape (n_samples,) or (n_samples, n_classes)
            Element-wise gradients.
        r  )r	  rN   s         r,   rM   zArrayAPILossMixin.gradientq  s*   € ð< ×%Ò%ØØ)Ø'ð &ñ 
ô 
ð 	
r.   ry   rw   rx   )r{   r|   r}   r~   rV   rE   rK   rM   r€   r.   r,   r  r  ñ  s¡   € € € € € ðð ð" ØðKð Kð Kð KðF ØØð!
ð !
ð !
ð !
ðN ØØØð.ð .ð .ð .ðh ØØð"
ð "
ð "
ð "
ð "
ð "
r.   r  c                 óŒ  — | j         |j        k    rg d¢ng d¢}|                     | |d         k    ||                     | |d         k    |                     |¦  «        |                     | |d         k    |                     d|z   ¦  «        |                     | |d         k    | d|z  z   | ¦  «        ¦  «        ¦  «        ¦  «        S )an  Numerically stable version of log(1 + exp(x)) that is compatible with
    the array API.

    Parameters
    ----------
    raw_prediction : C-contiguous array of shape (n_samples,) or array of         shape (n_samples, n_classes)
        Raw prediction values (in link space).
    raw_prediction_exp : C-contiguous array of shape (n_samples,) or array of         shape (n_samples, n_classes)
        Exponential of the raw prediction values.
    xp : module, default=None
        Array namespace module.

    Returns
    -------
    log1pexp : float
        Numerically stable value for log(1 + exp(raw_prediction)).
    )éÛÿÿÿrô   é   gfffff¦@@)éïÿÿÿrå   é	   g333333-@r   r8   r:   g      ð?rÚ   )rI   rn   ÚwhereÚlog1prº   )r=   Úraw_prediction_expr#   Ú	constantss       r,   Ú	_log1pexpr  –  så   € ðj Ô 2¤:Ò-Ð-ð 	ÐÐÐàÐÐð ð
 �8Š8Ø˜) Aœ,Ò&ØØ
�ŠØ˜i¨œlÒ*Ø�HŠHÐ'Ñ(Ô(Ø�HŠHØ )¨A¤,Ò.Ø—’�sÐ/Ñ/Ñ0Ô0Ø—’Ø" i°¤lÒ2Ø" QÐ);Ñ%;Ñ;Ø"ñô ñô ñ	
ô 	
ñô ð r.   c                   ó:   — e Zd ZdZ	 	 	 	 dd„Z	 	 dd„Z	 	 dd„ZdS )	ÚHalfBinomialLossArrayAPIzHA version of the HalfBinomialLoss that is compatible with the array API.Nr8   c                 ó¢   — | j                              |¦  «        }|                      ||||¬¦  «        }|                      ||||¬¦  «        }	||	fS ©N)r<   r=   r>   r  ©r#   Úexpr  r	  ©
r+   r<   r=   r>   r?   rJ   r@   r  rE   rM   s
             r,   rK   z&HalfBinomialLossArrayAPI.loss_gradientä  ór   € ð "œWŸ[š[¨Ñ8Ô8ÐØ×!Ò!ØØ)Ø'Ø1ð	 "ñ 
ô 
ˆð ×)Ò)ØØ)Ø'Ø1ð	 *ñ 
ô 
ˆð �Xˆ~Ðr.   c                 óŠ   — |€| j                              |¦  «        }t          ||| j         ¬¦  «        }|||z  z
  }|�||z  }|S )N)r=   r  r#   )r#   r  r  )r+   r<   r=   r>   r  Úlog1pexprE   s          r,   r  z&HalfBinomialLossArrayAPI._compute_lossü  sc   € ð Ð%Ø!%¤§¢¨^Ñ!<Ô!<ÐÝØ)Ø1ØŒwð
ñ 
ô 
ˆð
 ˜& >Ñ1Ñ1ˆØÐ$Ø�MÑ!ˆDØˆr.   c                 óØ   — | j         }|€|                     |¦  «        }d|z  }|                     ||j        |j        k    rdndk    d|z
  ||z  z
  d|z   z  ||z
  ¦  «        }|�||z  }|S )Nr8   r  r  )r#   r  r  rI   rn   )r+   r<   r=   r>   r  r#   Úneg_raw_prediction_expÚgrads           r,   r	  z*HalfBinomialLossArrayAPI._compute_gradient  sœ   € ð ŒWˆØÐ%Ø!#§¢¨Ñ!7Ô!7ÐØ!"Ð%7Ñ!7ÐØ�xŠxØ ^Ô%9¸R¼ZÒ%GÐ%G˜c˜cÈSÒQØ�&‰j˜FÐ%;Ñ;Ñ;ØÐ)Ñ)ñ+à Ñ'ñ	
ô 
ˆð Ð$Ø�MÑ!ˆDØˆr.   rx   ©NN©r{   r|   r}   r~   rK   r  r	  r€   r.   r,   r  r  á  st   € € € € € ØRÐRð ØØØðð ð ð ð8 Øðð ð ð ð. Øðð ð ð ð ð r.   r  c                   ó8   ‡ — e Zd ZdZdˆ fd„	Z	 dd„Z	 dd„Zˆ xZS )	ÚHalfMultinomialLossArrayAPIa�  A version of the HalfMultinomialLoss that is compatible with the array API.

    Parameters
    ----------
    sample_weight : {None, ndarray}
        If sample_weight is None, the hessian might be constant.

    n_classes : {None, int}
        The number of classes for classification, else None.

    xp : module or None
        Array namespace module.

    device : device or None
        A device object.
    NrÚ   c                 óz   •— t          ¦   «                              |||¬¦  «         d | _        d | _        d | _        d S )N)r"   r#   r$   )r…   r-   rÜ   rÝ   rÞ   rß   s        €r,   r-   z$HalfMultinomialLossArrayAPI.__init__7  sC   ø€ Ý‰Œ×Ò 9°¸FÐÑCÔCÐCð '+ˆÔ#ØˆŒð #ˆÔÐÐr.   c                 ó|  — | j         }| j        }t          |d|¬¦  «        }| j        €"|                     ||j        |¬¦  «        | _        | j        €/|                     |j        d         |¬¦  «        | j	        z  | _        | 
                    t          |¦  «        | j        | j        z   ¦  «        }||z
  }|�||z  }|S )Nr8   )rY   r#   ©rI   r$   r   )r$   )r#   r$   r   rÝ   ÚasarrayÚint64rÜ   ÚarangerC   r"   Útaker   )	r+   r<   r=   r>   r#   r$   Úlog_sum_expÚtrue_label_probsrE   s	            r,   r  z)HalfMultinomialLossArrayAPI._compute_lossD  sÉ   € ð ŒWˆØ”ˆÝ  °a¸BÐ?Ñ?Ô?ˆØŒ?Ð"Ø Ÿjšj¨°r´xÈ˜jÑOÔOˆDŒOàÔ&Ð.à—	’	˜&œ, qœ/°&�	Ñ9Ô9¸D¼NÑJð Ô'ð Ÿ7š7Ý�>Ñ"Ô" D¤O°dÔ6QÑ$Qñ
ô 
Ðð Ð-Ñ-ˆØÐ$Ø�MÑ!ˆDØˆr.   c                 ó"  — | j         }| j        }| j        €O| j        €"|                     ||j        |¬¦  «        | _        t          | j        | j        |j        ¬¦  «        | _        t          |¦  «        }|| j        z  }|�||d d …d f         z  }|S )Nr(  )Únum_classesrI   )
r#   r$   rÞ   rÝ   r)  r*  r   r"   rI   r   )r+   r<   r=   r>   r#   Údevice_r!  s          r,   r	  z-HalfMultinomialLossArrayAPI._compute_gradient\  s¥   € ð ŒWˆØ”+ˆØÔÐ&ØŒÐ&Ø"$§*¢*¨V¸2¼8ÈG *Ñ"TÔ"T�”å")Ø”Ø œNØ$Ô*ð#ñ #ô #ˆDÔõ
 �~Ñ&Ô&ˆð 	�Ô#Ñ#ˆØÐ$Ø�M ! ! ! T 'Ô*Ñ*ˆDØˆr.   rï   rz   )r{   r|   r}   r~   r-   r  r	  r‰   rŠ   s   @r,   r%  r%  %  sy   ø€ € € € € ðð ð"#ð #ð #ð #ð #ð #ð" ð	ð ð ð ð8 ð	ð ð ð ð ð ð ð r.   r%  c                   ó:   — e Zd ZdZ	 	 	 	 dd„Z	 	 dd„Z	 	 dd„ZdS )	ÚHalfPoissonLossArrayAPIzGA version of the HalfPoissonLoss that is compatible with the array API.Nr8   c                 ó¢   — | j                              |¦  «        }|                      ||||¬¦  «        }|                      ||||¬¦  «        }	||	fS r  r  r  s
             r,   rK   z%HalfPoissonLossArrayAPI.loss_gradient€  r  r.   c                 ó\   — |€| j                              |¦  «        }|||z  z
  }|�||z  }|S rz   ©r#   r  )r+   r<   r=   r>   r  rE   s         r,   r  z%HalfPoissonLossArrayAPI._compute_loss˜  sB   € ð Ð%Ø!%¤§¢¨^Ñ!<Ô!<ÐØ! F¨^Ñ$;Ñ;ˆØÐ$Ø�MÑ!ˆDØˆr.   c                 óV   — |€| j                              |¦  «        }||z
  }|�||z  }|S rz   r6  )r+   r<   r=   r>   r  r!  s         r,   r	  z)HalfPoissonLossArrayAPI._compute_gradient¦  s=   € ð Ð%Ø!%¤§¢¨^Ñ!<Ô!<ÐØ! FÑ*ˆØÐ$Ø�MÑ!ˆDØˆr.   rx   r"  r#  r€   r.   r,   r3  r3  }  st   € € € € € ØQÐQð ØØØðð ð ð ð8 Øðð ð ð ð$ Øðð ð ð ð ð r.   r3  )7r~   rŸ   Únumpyr'   Úscipy.specialr   Úsklearn._loss._lossr   r   r   r   r	   r
   r   r   r   r   r   Úsklearn._loss.linkr   r   r   r   r   r   Ú!sklearn.externals.array_api_extrar   Úsklearn.utilsr   Úsklearn.utils._array_apir   r   r   Úsklearn.utils.extmathr   Úsklearn.utils.statsr   r   r‚   rŒ   r•   r¦   r±   r·   r¼   rÇ   rÊ   rÙ   rñ   Ú_LOSSESr  r  r  r%  r3  r€   r.   r,   ú<module>rB     sÝ  ððð ð* €€€à Ð Ð Ð Ø Ð Ð Ð Ð Ð ðð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ðð ð ð ð ð ð ð ð ð ð ð ð ð ð ð ð 6Ð 5Ð 5Ð 5Ð 5Ð 5Ø &Ð &Ð &Ð &Ð &Ð &ðð ð ð ð ð ð ð ð ð ð
 *Ð )Ð )Ð )Ð )Ð )Ø 4Ð 4Ð 4Ð 4Ð 4Ð 4ð*T!ð T!ð T!ð T!ð T!ñ T!ô T!ð T!ðn6ð 6ð 6ð 6ð 6�xñ 6ô 6ð 6ð2$Cð $Cð $Cð $Cð $C�Hñ $Cô $Cð $CðN=ð =ð =ð =ð =�(ñ =ô =ð =ð@I@ð I@ð I@ð I@ð I@�ñ I@ô I@ð I@ðXð ð ð ð �hñ ô ð ðDð ð ð ð �Hñ ô ð ð>?ð ?ð ?ð ?ð ?�hñ ?ô ?ð ?ðD-Eð -Eð -Eð -Eð -E˜hñ -Eô -Eð -Eð`Dð Dð Dð Dð D�xñ Dô Dð DðNk'ð k'ð k'ð k'ð k'˜(ñ k'ô k'ð k'ð\Hð Hð Hð Hð H�hñ Hô Hð HðX &Ø#ØØØ#ØØ#Ø%Ø+Ø'ðð €ðb
ð b
ð b
ð b
ð b
ñ b
ô b
ð b
ðJHð Hð HðVAð Að Að Að AÐ0Ð2Bñ Aô Að AðHUð Uð Uð Uð UÐ"3Ð5Hñ Uô Uð Uðp5ð 5ð 5ð 5ð 5Ð/°ñ 5ô 5ð 5ð 5ð 5r.   