Ë
    ÜÍ:jÃæ  ã                   ó>  — d Z ddlZddlZddlmZ ddlmZmZm	Z	m
Z
mZmZmZmZmZmZmZ ddlmZmZmZmZmZmZ ddlmZ ddlmZ ddlmZmZm Z  dd	l!m"Z" dd
l#m$Z$  G d„ d«      Z% G d„ de%«      Z& G d„ de%«      Z' G d„ de%«      Z( G d„ de%«      Z) G d„ de%«      Z* G d„ de%«      Z+ G d„ de%«      Z, G d„ de%«      Z- G d„ de%«      Z. G d„ d e%«      Z/ G d!„ d"e%«      Z0e&e'e(e)e*e+e,e.e/e0d#œ
Z1 G d$„ d%«      Z2d&„ Z3 G d'„ d(e2e.«      Z4 G d)„ d*e2e/«      Z5 G d+„ d,e2e*«      Z6y)-z¹
This module contains loss classes suitable for fitting.

It is not part of the public API.
Specific losses are used for regression, binary classification or multiclass
classification.
é    N©Úxlogy)ÚCyAbsoluteErrorÚCyExponentialLossÚCyHalfBinomialLossÚCyHalfGammaLossÚCyHalfMultinomialLossÚCyHalfPoissonLossÚCyHalfSquaredErrorÚCyHalfTweedieLossÚCyHalfTweedieLossIdentityÚCyHuberLossÚCyPinballLoss)ÚHalfLogitLinkÚIdentityLinkÚIntervalÚ	LogitLinkÚLogLinkÚMultinomialLogit)Úone_hot)Úcheck_scalar)Ú_averageÚ
_logsumexpÚ_ravel)Úsoftmax)Ú_weighted_percentilec                   ó    — e Zd ZdZdZdZdd„Zd„ Zd„ Z	 	 	 dd„Z		 	 	 	 dd	„Z
	 	 	 dd
„Z	 	 	 	 dd„Zdd„Zdd„Zdd„Zej"                  dfd„Zy)ÚBaseLossaÒ  Base class for a loss function of 1-dimensional targets.

    Conventions:

        - y_true.shape = sample_weight.shape = (n_samples,)
        - y_pred.shape = raw_prediction.shape = (n_samples,)
        - If is_multiclass is true (multiclass classification), then
          y_pred.shape = raw_prediction.shape = (n_samples, n_classes)
          Note that this corresponds to the return value of decision_function.

    y_true, y_pred, sample_weight and raw_prediction must either be all float64
    or all float32.
    gradient and hessian must be either both float64 or both float32.

    Note that y_pred = link.inverse(raw_prediction).

    Specific loss classes can inherit specific link classes to satisfy
    BaseLink's abstractmethods.

    Parameters
    ----------
    closs: CyLossFunction
        For example, a CyLossFunction; hence the name "c"loss.
    link : BaseLink
    sample_weight : {None, ndarray}
        If sample_weight is None, the hessian might be constant.
    n_classes : {None, int}
        The number of classes for classification, else None.
    xp : module, default=None
        Array namespace module.
    device : device, default=None
        A device object (see the "Device Support" section of the array API spec).

    Attributes
    ----------
    closs: CyLossFunction
        For example, a CyLossFunction; hence the name "c"loss.
    link : BaseLink
    n_classes : {None, int}
        The number of classes for classification, else None.
    xp : module or None
        Array namespace module. Ignored by the Cython implementation.
    device : device or None
        A device object. Ignored by the Cython implementation.
    interval_y_true : Interval
        Valid interval for y_true
    interval_y_pred : Interval
        Valid Interval for y_pred
    differentiable : bool
        Indicates whether or not loss function is differentiable in
        raw_prediction everywhere.
    approx_hessian : bool
        Indicates whether the hessian is approximated or exact. If,
        approximated, it should be larger or equal to the exact one.
    constant_hessian : bool
        Indicates whether the hessian is one for this loss.
    is_multiclass : bool
        Indicates whether n_classes > 2 is allowed.
    TFNc                 óü   — || _         || _        || _        || _        || _        d| _        d| _        t        t        j                   t        j                  dd«      | _
        | j                  j                  | _        y )NF)ÚclossÚlinkÚ	n_classesÚxpÚdeviceÚapprox_hessianÚconstant_hessianr   ÚnpÚinfÚinterval_y_trueÚinterval_y_pred)Úselfr    r!   r"   r#   r$   s         úg/home/mcse/projects/srt_converter/srt-converter-venv/lib/python3.12/site-packages/sklearn/_loss/loss.pyÚ__init__zBaseLoss.__init__–   sd   € ØˆŒ
ØˆŒ	Ø"ˆŒØˆŒØˆŒØ#ˆÔØ %ˆÔÜ'¬¯©¨´·±¸ÀÓFˆÔØ#Ÿy™y×8Ñ8ˆÕó    c                 ó8   — | j                   j                  |«      S ©zuReturn True if y is in the valid range of y_true.

        Parameters
        ----------
        y : ndarray
        )r)   Úincludes©r+   Úys     r,   Úin_y_true_rangezBaseLoss.in_y_true_range¡   ó   € ð ×#Ñ#×,Ñ,¨QÓ/Ð/r.   c                 ó8   — | j                   j                  |«      S )zuReturn True if y is in the valid range of y_pred.

        Parameters
        ----------
        y : ndarray
        )r*   r1   r2   s     r,   Úin_y_pred_rangezBaseLoss.in_y_pred_rangeª   r5   r.   c                 óØ   — |€t        j                  |«      }|j                  dk(  r#|j                  d   dk(  r|j	                  d«      }| j
                  j                  |||||¬«       |S )aJ  Compute the pointwise loss value for each input.

        Parameters
        ----------
        y_true : C-contiguous array of shape (n_samples,)
            Observed, true target values.
        raw_prediction : C-contiguous array of shape (n_samples,) or array of             shape (n_samples, n_classes)
            Raw prediction values (in link space).
        sample_weight : None or C-contiguous array of shape (n_samples,)
            Sample weights.
        loss_out : None or C-contiguous array of shape (n_samples,)
            A location into which the result is stored. If None, a new array
            might be created.
        n_threads : int, default=1
            Might use openmp thread parallelism.

        Returns
        -------
        loss : array of shape (n_samples,)
            Element-wise loss function.
        é   é   ©Úy_trueÚraw_predictionÚsample_weightÚloss_outÚ	n_threads)r'   Ú
empty_likeÚndimÚshapeÚsqueezer    Úloss©r+   r<   r=   r>   r?   r@   s         r,   rE   zBaseLoss.loss³   ss   € ð< ÐÜ—}‘} VÓ,ˆHà×Ñ !Ò#¨×(<Ñ(<¸QÑ(?À1Ò(DØ+×3Ñ3°AÓ6ˆNà�
‰
�‰ØØ)Ø'ØØð 	ô 	
ð ˆr.   c                 óü  — |€O|€+t        j                  |«      }t        j                  |«      }nEt        j                  ||j                  ¬«      }n#|€!t        j                  ||j                  ¬«      }|j                  dk(  r#|j                  d   dk(  r|j                  d«      }|j                  dk(  r#|j                  d   dk(  r|j                  d«      }| j                  j                  ||||||¬«       ||fS )a¯  Compute loss and gradient w.r.t. raw_prediction for each input.

        Parameters
        ----------
        y_true : C-contiguous array of shape (n_samples,)
            Observed, true target values.
        raw_prediction : C-contiguous array of shape (n_samples,) or array of             shape (n_samples, n_classes)
            Raw prediction values (in link space).
        sample_weight : None or C-contiguous array of shape (n_samples,)
            Sample weights.
        loss_out : None or C-contiguous array of shape (n_samples,)
            A location into which the loss is stored. If None, a new array
            might be created.
        gradient_out : None or C-contiguous array of shape (n_samples,) or array             of shape (n_samples, n_classes)
            A location into which the gradient is stored. If None, a new array
            might be created.
        n_threads : int, default=1
            Might use openmp thread parallelism.

        Returns
        -------
        loss : array of shape (n_samples,)
            Element-wise loss function.

        gradient : array of shape (n_samples,) or (n_samples, n_classes)
            Element-wise gradients.
        ©Údtyper9   r:   )r<   r=   r>   r?   Úgradient_outr@   )r'   rA   rI   rB   rC   rD   r    Úloss_gradient)r+   r<   r=   r>   r?   rJ   r@   s          r,   rK   zBaseLoss.loss_gradientà   sõ   € ðL ÐØÐ#ÜŸ=™=¨Ó0�Ü!Ÿ}™}¨^Ó<‘äŸ=™=¨°|×7IÑ7IÔJ‘ØÐ!ÜŸ=™=¨¸x¿~¹~ÔNˆLð ×Ñ !Ò#¨×(<Ñ(<¸QÑ(?À1Ò(DØ+×3Ñ3°AÓ6ˆNØ×Ñ Ò! l×&8Ñ&8¸Ñ&;¸qÒ&@Ø'×/Ñ/°Ó2ˆLà�
‰
× Ñ ØØ)Ø'ØØ%Øð 	!ô 	
ð ˜Ð%Ð%r.   c                 ó<  — |€t        j                  |«      }|j                  dk(  r#|j                  d   dk(  r|j	                  d«      }|j                  dk(  r#|j                  d   dk(  r|j	                  d«      }| j
                  j                  |||||¬«       |S )aª  Compute gradient of loss w.r.t raw_prediction for each input.

        Parameters
        ----------
        y_true : C-contiguous array of shape (n_samples,)
            Observed, true target values.
        raw_prediction : C-contiguous array of shape (n_samples,) or array of             shape (n_samples, n_classes)
            Raw prediction values (in link space).
        sample_weight : None or C-contiguous array of shape (n_samples,)
            Sample weights.
        gradient_out : None or C-contiguous array of shape (n_samples,) or array             of shape (n_samples, n_classes)
            A location into which the result is stored. If None, a new array
            might be created.
        n_threads : int, default=1
            Might use openmp thread parallelism.

        Returns
        -------
        gradient : array of shape (n_samples,) or (n_samples, n_classes)
            Element-wise gradients.
        r9   r:   )r<   r=   r>   rJ   r@   )r'   rA   rB   rC   rD   r    Úgradient©r+   r<   r=   r>   rJ   r@   s         r,   rM   zBaseLoss.gradient  s¨   € ð> ÐÜŸ=™=¨Ó8ˆLð ×Ñ !Ò#¨×(<Ñ(<¸QÑ(?À1Ò(DØ+×3Ñ3°AÓ6ˆNØ×Ñ Ò! l×&8Ñ&8¸Ñ&;¸qÒ&@Ø'×/Ñ/°Ó2ˆLà�
‰
×ÑØØ)Ø'Ø%Øð 	ô 	
ð Ðr.   c                 ó0  — |€C|€+t        j                  |«      }t        j                  |«      }n-t        j                  |«      }n|€t        j                  |«      }|j                  dk(  r#|j                  d   dk(  r|j	                  d«      }|j                  dk(  r#|j                  d   dk(  r|j	                  d«      }|j                  dk(  r#|j                  d   dk(  r|j	                  d«      }| j
                  j                  ||||||¬«       ||fS )aÿ  Compute gradient and hessian of loss w.r.t raw_prediction.

        Parameters
        ----------
        y_true : C-contiguous array of shape (n_samples,)
            Observed, true target values.
        raw_prediction : C-contiguous array of shape (n_samples,) or array of             shape (n_samples, n_classes)
            Raw prediction values (in link space).
        sample_weight : None or C-contiguous array of shape (n_samples,)
            Sample weights.
        gradient_out : None or C-contiguous array of shape (n_samples,) or array             of shape (n_samples, n_classes)
            A location into which the gradient is stored. If None, a new array
            might be created.
        hessian_out : None or C-contiguous array of shape (n_samples,) or array             of shape (n_samples, n_classes)
            A location into which the hessian is stored. If None, a new array
            might be created.
        n_threads : int, default=1
            Might use openmp thread parallelism.

        Returns
        -------
        gradient : arrays of shape (n_samples,) or (n_samples, n_classes)
            Element-wise gradients.

        hessian : arrays of shape (n_samples,) or (n_samples, n_classes)
            Element-wise hessians.
        r9   r:   )r<   r=   r>   rJ   Úhessian_outr@   )r'   rA   rB   rC   rD   r    Úgradient_hessian)r+   r<   r=   r>   rJ   rP   r@   s          r,   rQ   zBaseLoss.gradient_hessianP  s  € ðN ÐØÐ"Ü!Ÿ}™}¨^Ó<�Ü Ÿm™m¨NÓ;‘ä!Ÿ}™}¨[Ó9‘ØÐ ÜŸ-™-¨Ó5ˆKð ×Ñ !Ò#¨×(<Ñ(<¸QÑ(?À1Ò(DØ+×3Ñ3°AÓ6ˆNØ×Ñ Ò! l×&8Ñ&8¸Ñ&;¸qÒ&@Ø'×/Ñ/°Ó2ˆLØ×Ñ˜qÒ  [×%6Ñ%6°qÑ%9¸QÒ%>Ø%×-Ñ-¨aÓ0ˆKà�
‰
×#Ñ#ØØ)Ø'Ø%Ø#Øð 	$ô 	
ð ˜[Ð(Ð(r.   c           	      óX   — t        j                  | j                  ||dd|¬«      |¬«      S )a{  Compute the weighted average loss.

        Parameters
        ----------
        y_true : C-contiguous array of shape (n_samples,)
            Observed, true target values.
        raw_prediction : C-contiguous array of shape (n_samples,) or array of             shape (n_samples, n_classes)
            Raw prediction values (in link space).
        sample_weight : None or C-contiguous array of shape (n_samples,)
            Sample weights.
        n_threads : int, default=1
            Might use openmp thread parallelism.

        Returns
        -------
        loss : float
            Mean or averaged loss function.
        Nr;   ©Úweights)r'   ÚaveragerE   )r+   r<   r=   r>   r@   s        r,   Ú__call__zBaseLoss.__call__’  s:   € ô( �z‰zØ�I‰IØØ-Ø"ØØ#ð ó ð "ô	
ð 		
r.   c                 óê  — t        j                  ||d¬«      }dt        j                  |j                  «      j                  z  }| j
                  j                  t         j                   k(  rd}nF| j
                  j                  r| j
                  j                  }n| j
                  j                  |z   }| j
                  j                  t         j                  k(  rd}nF| j
                  j                  r| j
                  j                  }n| j
                  j                  |z
  }|€|€| j                  j                  |«      S | j                  j                  t        j                  |||«      «      S )a#  Compute raw_prediction of an intercept-only model.

        This can be used as initial estimates of predictions, i.e. before the
        first iteration in fit.

        Parameters
        ----------
        y_true : array-like of shape (n_samples,)
            Observed, true target values.
        sample_weight : None or array of shape (n_samples,)
            Sample weights.

        Returns
        -------
        raw_prediction : numpy scalar or array of shape (n_classes,)
            Raw predictions of an intercept-only model.
        r   ©rT   Úaxisé
   N)r'   rU   ÚfinforI   Úepsr*   Úlowr(   Úlow_inclusiveÚhighÚhigh_inclusiver!   Úclip)r+   r<   r>   Úy_predr\   Úa_minÚa_maxs          r,   Úfit_intercept_onlyzBaseLoss.fit_intercept_only±  s  € ô( —‘˜F¨MÀÔBˆØ”2—8‘8˜FŸL™LÓ)×-Ñ-Ñ-ˆà×Ñ×#Ñ#¬¯© wÒ.Ø‰EØ×!Ñ!×/Ò/Ø×(Ñ(×,Ñ,‰Eà×(Ñ(×,Ñ,¨sÑ2ˆEà×Ñ×$Ñ$¬¯©Ò.Ø‰EØ×!Ñ!×0Ò0Ø×(Ñ(×-Ñ-‰Eà×(Ñ(×-Ñ-°Ñ3ˆEàˆ=˜U˜]Ø—9‘9—>‘> &Ó)Ð)à—9‘9—>‘>¤"§'¡'¨&°%¸Ó"?Ó@Ð@r.   c                 ó,   — t        j                  |«      S )a(  Calculate term dropped in loss.

        With this term added, the loss of perfect predictions is zero.

        Parameters
        ----------
        y_true : array-like of shape (n_samples,)
            Observed, true target values.

        sample_weight : None or array of shape (n_samples,), default=None
            Sample weights.

        Returns
        -------
        constant : ndarray of shape (n_samples,)
            Constant value to be added to raw predictions so that the loss
            of perfect predictions becomes zero.
        )r'   Ú
zeros_like©r+   r<   r>   s      r,   Úconstant_to_optimal_zeroz!BaseLoss.constant_to_optimal_zeroÛ  s   € ô& �}‰}˜VÓ$Ð$r.   ÚFc                 óV  — |t         j                  t         j                  fvrt        d|› d�«      ‚| j                  r|| j
                  f}n|f}t        j                  |||¬«      }| j                  rt        j                  d|¬«      }||fS t        j                  |||¬«      }||fS )au  Initialize arrays for gradients and hessians.

        Unless hessians are constant, arrays are initialized with undefined values.

        Parameters
        ----------
        n_samples : int
            The number of samples, usually passed to `fit()`.
        dtype : {np.float64, np.float32}, default=np.float64
            The dtype of the arrays gradient and hessian.
        order : {'C', 'F'}, default='F'
            Order of the arrays gradient and hessian. The default 'F' makes the arrays
            contiguous along samples.

        Returns
        -------
        gradient : C-contiguous array of shape (n_samples,) or array of shape             (n_samples, n_classes)
            Empty array (allocated but not initialized) to be used as argument
            gradient_out.
        hessian : C-contiguous array of shape (n_samples,), array of shape
            (n_samples, n_classes) or shape (1,)
            Empty (allocated but not initialized) array to be used as argument
            hessian_out.
            If constant_hessian is True (e.g. `HalfSquaredError`), the array is
            initialized to ``1``.
        zCValid options for 'dtype' are np.float32 and np.float64. Got dtype=z	 instead.)rC   rI   Úorder)r:   )rC   rI   )	r'   Úfloat32Úfloat64Ú
ValueErrorÚis_multiclassr"   Úemptyr&   Úones)r+   Ú	n_samplesrI   rl   rC   rM   Úhessians          r,   Úinit_gradient_and_hessianz"BaseLoss.init_gradient_and_hessianð  s°   € ð8 œŸ™¤R§Z¡ZÐ0Ñ0ÜðØ"˜G 9ð.óð ð
 ×ÒØ §¡Ð/‰Eà�LˆEÜ—8‘8 %¨u¸EÔBˆà× Ò ô
 —g‘g D°Ô6ˆGð ˜Ð Ð ô —h‘h U°%¸uÔEˆGà˜Ð Ð r.   ©NNN©NNr:   ©NNNr:   ©Nr:   ©N)Ú__name__Ú
__module__Ú__qualname__Ú__doc__Údifferentiablerp   r-   r4   r7   rE   rK   rM   rQ   rV   re   ri   r'   rn   ru   © r.   r,   r   r   M   s�   „ ñ:ðJ €NØ€Mó	9ò0ò0ð ØØó+ðb ØØØó=&ðF ØØó/ðj ØØØó@)óD
ó>(AóT%ð* :<¿¹È3ô 1!r.   r   c                   ó$   ‡ — e Zd ZdZdˆ fd„	Zˆ xZS )ÚHalfSquaredErroraÜ  Half squared error with identity link, for regression.

    Domain:
    y_true and y_pred all real numbers

    Link:
    y_pred = raw_prediction

    For a given sample x_i, half squared error is defined as::

        loss(x_i) = 0.5 * (y_true_i - raw_prediction_i)**2

    The factor of 0.5 simplifies the computation of gradients and results in a
    unit hessian (and is consistent with what is done in LightGBM). It is also
    half the Normal distribution deviance.
    c                 ó^   •— t         ‰| �  t        «       t        «       ||¬«       |d u | _        y )N©r    r!   r#   r$   )Úsuperr-   r   r   r&   ©r+   r>   r#   r$   Ú	__class__s       €r,   r-   zHalfSquaredError.__init__6  s2   ø€ Ü‰ÑÜ$Ó&¬\«^ÀÈ6ð 	ô 	
ð !.°Ð 5ˆÕr.   rv   ©r{   r|   r}   r~   r-   Ú__classcell__©r‡   s   @r,   r‚   r‚   $  s   ø„ ñ÷"6ñ 6r.   r‚   c                   ó0   ‡ — e Zd ZdZdZdˆ fd„	Zdd„Zˆ xZS )ÚAbsoluteErroraÖ  Absolute error with identity link, for regression.

    Domain:
    y_true and y_pred all real numbers

    Link:
    y_pred = raw_prediction

    For a given sample x_i, the absolute error is defined as::

        loss(x_i) = |y_true_i - raw_prediction_i|

    Note that the exact hessian = 0 almost everywhere (except at one point, therefore
    differentiable = False). Optimization routines like in HGBT, however, need a
    hessian > 0. Therefore, we assign 1.
    Fc                 ól   •— t         ‰| �  t        «       t        «       ||¬«       d| _        |d u | _        y )Nr„   T)r…   r-   r   r   r%   r&   r†   s       €r,   r-   zAbsoluteError.__init__Q  s:   ø€ Ü‰ÑÜ!Ó#¬,«.¸RÈð 	ô 	
ð #ˆÔØ -°Ð 5ˆÕr.   c                 óN   — |€t        j                  |d¬«      S t        ||d«      S )ú•Compute raw_prediction of an intercept-only model.

        This is the weighted median of the target, i.e. over the samples
        axis=0.
        r   ©rY   é2   )r'   Úmedianr   rh   s      r,   re   z AbsoluteError.fit_intercept_onlyX  s*   € ð Ð Ü—9‘9˜V¨!Ô,Ð,ä'¨°¸rÓBÐBr.   rv   rz   ©r{   r|   r}   r~   r   r-   re   r‰   rŠ   s   @r,   rŒ   rŒ   =  s   ø„ ñð" €Nõ6÷	Cr.   rŒ   c                   ó0   ‡ — e Zd ZdZdZdˆ fd„	Zdd„Zˆ xZS )ÚPinballLossa  Quantile loss aka pinball loss, for regression.

    Domain:
    y_true and y_pred all real numbers
    quantile in (0, 1)

    Link:
    y_pred = raw_prediction

    For a given sample x_i, the pinball loss is defined as::

        loss(x_i) = rho_{quantile}(y_true_i - raw_prediction_i)

        rho_{quantile}(u) = u * (quantile - 1_{u<0})
                          = -u *(1 - quantile)  if u < 0
                             u * quantile       if u >= 0

    Note: 2 * PinballLoss(quantile=0.5) equals AbsoluteError().

    Note that the exact hessian = 0 almost everywhere (except at one point, therefore
    differentiable = False). Optimization routines like in HGBT, however, need a
    hessian > 0. Therefore, we assign 1.

    Additional Attributes
    ---------------------
    quantile : float
        The quantile level of the quantile to be estimated. Must be in range (0, 1).
    Fc                 óÀ   •— t        |dt        j                  ddd¬«       t        ‰| �  t        t        |«      ¬«      t        «       ||¬«       d| _        |d u | _	        y )	NÚquantiler   r:   Úneither©Útarget_typeÚmin_valÚmax_valÚinclude_boundaries)r—   r„   T)
r   ÚnumbersÚRealr…   r-   r   Úfloatr   r%   r&   )r+   r>   r—   r#   r$   r‡   s        €r,   r-   zPinballLoss.__init__„  sc   ø€ ÜØØÜŸ™ØØØ(õ	
ô 	‰ÑÜ¬¨x«Ô9Ü“ØØð	 	ô 	
ð #ˆÔØ -°Ð 5ˆÕr.   c                 ó¬   — |€/t        j                  |d| j                  j                  z  d¬«      S t	        ||d| j                  j                  z  «      S )r�   éd   r   r�   )r'   Ú
percentiler    r—   r   rh   s      r,   re   zPinballLoss.fit_intercept_only–  sO   € ð Ð Ü—=‘= ¨¨t¯z©z×/BÑ/BÑ)BÈÔKÐKä'Ø˜ s¨T¯Z©Z×-@Ñ-@Ñ'@óð r.   )Nç      à?NNrz   r“   rŠ   s   @r,   r•   r•   d  s   ø„ ñð: €Nõ6÷$r.   r•   c                   ó2   ‡ — e Zd ZdZdZ	 dˆ fd„	Zdd„Zˆ xZS )Ú	HuberLossaê  Huber loss, for regression.

    Domain:
    y_true and y_pred all real numbers
    quantile in (0, 1)

    Link:
    y_pred = raw_prediction

    For a given sample x_i, the Huber loss is defined as::

        loss(x_i) = 1/2 * abserr**2            if abserr <= delta
                    delta * (abserr - delta/2) if abserr > delta

        abserr = |y_true_i - raw_prediction_i|
        delta = quantile(abserr, self.quantile)

    Note: HuberLoss(quantile=1) equals HalfSquaredError and HuberLoss(quantile=0)
    equals delta * (AbsoluteError() - delta/2).

    Additional Attributes
    ---------------------
    quantile : float
        The quantile level which defines the breaking point `delta` to distinguish
        between absolute error and squared error. Must be in range (0, 1).

     Reference
    ---------
    .. [1] Friedman, J.H. (2001). :doi:`Greedy function approximation: A gradient
      boosting machine <10.1214/aos/1013203451>`.
      Annals of Statistics, 29, 1189-1232.
    Fc                 óÊ   •— t        |dt        j                  ddd¬«       || _        t        ‰| �  t        t        |«      ¬«      t        «       ||¬«       d| _	        d	| _
        y )
Nr—   r   r:   r˜   r™   )Údeltar„   TF)r   rž   rŸ   r—   r…   r-   r   r    r   r%   r&   )r+   r>   r—   r¨   r#   r$   r‡   s         €r,   r-   zHuberLoss.__init__È  sg   ø€ ô 	ØØÜŸ™ØØØ(õ	
ð !ˆŒÜ‰ÑÜ¤E¨%£LÔ1Ü“ØØð	 	ô 	
ð #ˆÔØ %ˆÕr.   c                 ó6  — |€t        j                  |dd¬«      }nt        ||d«      }||z
  }t        j                  |«      t        j                  | j
                  j                  t        j                  |«      «      z  }|t        j                  ||¬«      z   S )r�   r‘   r   r�   rS   )	r'   r£   r   ÚsignÚminimumr    r¨   ÚabsrU   )r+   r<   r>   r’   ÚdiffÚterms         r,   re   zHuberLoss.fit_intercept_onlyÝ  sx   € ð Ð Ü—]‘] 6¨2°AÔ6‰Fä)¨&°-ÀÓDˆFØ˜‰ˆÜ�w‰w�t‹}œrŸz™z¨$¯*©*×*:Ñ*:¼B¿F¹FÀ4»LÓIÑIˆØœŸ
™
 4°Ô?Ñ?Ð?r.   )NgÍÌÌÌÌÌì?r¤   NNrz   r“   rŠ   s   @r,   r¦   r¦   ¤  s"   ø„ ñðB €Nð LPõ&÷*@r.   r¦   c                   ó,   ‡ — e Zd ZdZdˆ fd„	Zdd„Zˆ xZS )ÚHalfPoissonLossaƒ  Half Poisson deviance loss with log-link, for regression.

    Domain:
    y_true in non-negative real numbers
    y_pred in positive real numbers

    Link:
    y_pred = exp(raw_prediction)

    For a given sample x_i, half the Poisson deviance is defined as::

        loss(x_i) = y_true_i * log(y_true_i/exp(raw_prediction_i))
                    - y_true_i + exp(raw_prediction_i)

    Half the Poisson deviance is actually the negative log-likelihood up to
    constant terms (not involving raw_prediction) and simplifies the
    computation of the gradients.
    We also skip the constant term `y_true_i * log(y_true_i) - y_true_i`.
    c                 óŽ   •— t         ‰| �  t        «       t        «       ||¬«       t	        dt
        j                  dd«      | _        y )Nr„   r   TF)r…   r-   r
   r   r   r'   r(   r)   r†   s       €r,   r-   zHalfPoissonLoss.__init__  s<   ø€ Ü‰ÑÜ#Ó%¬G«I¸"ÀVð 	ô 	
ô  (¨¬2¯6©6°4¸Ó?ˆÕr.   c                 ó2   — t        ||«      |z
  }|�||z  }|S rz   r   ©r+   r<   r>   r®   s       r,   ri   z(HalfPoissonLoss.constant_to_optimal_zero  s(   € Ü�V˜VÓ$ vÑ-ˆØÐ$Ø�MÑ!ˆDØˆr.   rv   rz   ©r{   r|   r}   r~   r-   ri   r‰   rŠ   s   @r,   r°   r°   ð  s   ø„ ñõ(@÷r.   r°   c                   ó,   ‡ — e Zd ZdZdˆ fd„	Zdd„Zˆ xZS )ÚHalfGammaLossaV  Half Gamma deviance loss with log-link, for regression.

    Domain:
    y_true and y_pred in positive real numbers

    Link:
    y_pred = exp(raw_prediction)

    For a given sample x_i, half Gamma deviance loss is defined as::

        loss(x_i) = log(exp(raw_prediction_i)/y_true_i)
                    + y_true/exp(raw_prediction_i) - 1

    Half the Gamma deviance is actually proportional to the negative log-
    likelihood up to constant terms (not involving raw_prediction) and
    simplifies the computation of the gradients.
    We also skip the constant term `-log(y_true_i) - 1`.
    c                 óŽ   •— t         ‰| �  t        «       t        «       ||¬«       t	        dt
        j                  dd«      | _        y )Nr„   r   F)r…   r-   r   r   r   r'   r(   r)   r†   s       €r,   r-   zHalfGammaLoss.__init__&  s6   ø€ Ü‰ÑœÓ0´w³yÀRÐPVÐÔWÜ'¨¬2¯6©6°5¸%Ó@ˆÕr.   c                 óF   — t        j                  |«       dz
  }|�||z  }|S ry   )r'   Úlogr³   s       r,   ri   z&HalfGammaLoss.constant_to_optimal_zero*  s+   € Ü—‘�v“ˆ Ñ"ˆØÐ$Ø�MÑ!ˆDØˆr.   rv   rz   r´   rŠ   s   @r,   r¶   r¶     s   ø„ ñõ&A÷r.   r¶   c                   ó,   ‡ — e Zd ZdZdˆ fd„	Zdd„Zˆ xZS )ÚHalfTweedieLossaÿ  Half Tweedie deviance loss with log-link, for regression.

    Domain:
    y_true in real numbers for power <= 0
    y_true in non-negative real numbers for 0 < power < 2
    y_true in positive real numbers for 2 <= power
    y_pred in positive real numbers
    power in real numbers

    Link:
    y_pred = exp(raw_prediction)

    For a given sample x_i, half Tweedie deviance loss with p=power is defined
    as::

        loss(x_i) = max(y_true_i, 0)**(2-p) / (1-p) / (2-p)
                    - y_true_i * exp(raw_prediction_i)**(1-p) / (1-p)
                    + exp(raw_prediction_i)**(2-p) / (2-p)

    Taking the limits for p=0, 1, 2 gives HalfSquaredError with a log link,
    HalfPoissonLoss and HalfGammaLoss.

    We also skip constant terms, but those are different for p=0, 1, 2.
    Therefore, the loss is not continuous in `power`.

    Note furthermore that although no Tweedie distribution exists for
    0 < power < 1, it still gives a strictly consistent scoring function for
    the expectation.
    c                 ó®  •— t         ‰| �  t        t        |«      ¬«      t	        «       ||¬«       | j
                  j                  dk  r1t        t        j                   t        j                  dd«      | _
        y | j
                  j                  dk  r"t        dt        j                  dd«      | _
        y t        dt        j                  dd«      | _
        y ©N)Úpowerr„   r   Fr9   T)r…   r-   r   r    r   r    r¾   r   r'   r(   r)   ©r+   r>   r¾   r#   r$   r‡   s        €r,   r-   zHalfTweedieLoss.__init__P  s�   ø€ Ü‰ÑÜ#¬%°«,Ô7Ü“ØØð	 	ô 	
ð �:‰:×Ñ˜qÒ Ü#+¬R¯V©V¨G´R·V±V¸UÀEÓ#JˆDÕ Ø�Z‰Z×Ñ Ò!Ü#+¨A¬r¯v©v°t¸UÓ#CˆDÕ ä#+¨A¬r¯v©v°u¸eÓ#DˆDÕ r.   c                 óê  — | j                   j                  dk(  rt        «       j                  ||¬«      S | j                   j                  dk(  rt	        «       j                  ||¬«      S | j                   j                  dk(  rt        «       j                  ||¬«      S | j                   j                  }t        j                  t        j                  |d«      d|z
  «      d|z
  z  d|z
  z  }|�||z  }|S )Nr   )r<   r>   r:   r9   )r    r¾   r‚   ri   r°   r¶   r'   Úmaximum)r+   r<   r>   Úpr®   s        r,   ri   z(HalfTweedieLoss.constant_to_optimal_zero^  sò   € Ø�:‰:×Ñ˜qÒ Ü#Ó%×>Ñ>Ø¨]ð ?ó ð ð �Z‰Z×Ñ Ò"Ü"Ó$×=Ñ=Ø¨]ð >ó ð ð �Z‰Z×Ñ Ò"Ü “?×;Ñ;Ø¨]ð <ó ð ð —
‘
× Ñ ˆAÜ—8‘8œBŸJ™J v¨qÓ1°1°q±5Ó9¸QÀ¹UÑCÀqÈ1ÁuÑMˆDØÐ(Ø˜Ñ%�ØˆKr.   ©Ng      ø?NNrz   r´   rŠ   s   @r,   r»   r»   1  s   ø„ ñõ<E÷r.   r»   c                   ó$   ‡ — e Zd ZdZdˆ fd„	Zˆ xZS )ÚHalfTweedieLossIdentityan  Half Tweedie deviance loss with identity link, for regression.

    Domain:
    y_true in real numbers for power <= 0
    y_true in non-negative real numbers for 0 < power < 2
    y_true in positive real numbers for 2 <= power
    y_pred in positive real numbers for power != 0
    y_pred in real numbers for power = 0
    power in real numbers

    Link:
    y_pred = raw_prediction

    For a given sample x_i, half Tweedie deviance loss with p=power is defined
    as::

        loss(x_i) = max(y_true_i, 0)**(2-p) / (1-p) / (2-p)
                    - y_true_i * raw_prediction_i**(1-p) / (1-p)
                    + raw_prediction_i**(2-p) / (2-p)

    Note that the minimum value of this loss is 0.

    Note furthermore that although no Tweedie distribution exists for
    0 < power < 1, it still gives a strictly consistent scoring function for
    the expectation.
    c                 ó„  •— t         ‰| �  t        t        |«      ¬«      t	        «       ||¬«       | j
                  j                  dk  r1t        t        j                   t        j                  dd«      | _
        n\| j
                  j                  dk  r"t        dt        j                  dd«      | _
        n!t        dt        j                  dd«      | _
        | j
                  j                  dk(  r1t        t        j                   t        j                  dd«      | _        y t        dt        j                  dd«      | _        y r½   )r…   r-   r   r    r   r    r¾   r   r'   r(   r)   r*   r¿   s        €r,   r-   z HalfTweedieLossIdentity.__init__�  sã   ø€ Ü‰ÑÜ+´%¸³,Ô?Ü“ØØð	 	ô 	
ð �:‰:×Ñ˜qÒ Ü#+¬R¯V©V¨G´R·V±V¸UÀEÓ#JˆDÕ Ø�Z‰Z×Ñ Ò!Ü#+¨A¬r¯v©v°t¸UÓ#CˆDÕ ä#+¨A¬r¯v©v°u¸eÓ#DˆDÔ à�:‰:×Ñ˜qÒ Ü#+¬R¯V©V¨G´R·V±V¸UÀEÓ#JˆDÕ ä#+¨A¬r¯v©v°u¸eÓ#DˆDÕ r.   rÃ   rˆ   rŠ   s   @r,   rÅ   rÅ   s  s   ø„ ñ÷6Eñ Er.   rÅ   c                   ó2   ‡ — e Zd ZdZdˆ fd„	Zdd„Zd„ Zˆ xZS )ÚHalfBinomialLossaY  Half Binomial deviance loss with logit link, for binary classification.

    This is also know as binary cross entropy, log-loss and logistic loss.

    Domain:
    y_true in [0, 1], i.e. regression on the unit interval
    y_pred in (0, 1), i.e. boundaries excluded

    Link:
    y_pred = expit(raw_prediction)

    For a given sample x_i, half Binomial deviance is defined as the negative
    log-likelihood of the Binomial/Bernoulli distribution and can be expressed
    as::

        loss(x_i) = log(1 + exp(raw_pred_i)) - y_true_i * raw_pred_i

    See The Elements of Statistical Learning, by Hastie, Tibshirani, Friedman,
    section 4.4.1 (about logistic regression).

    Note that the formulation works for classification, y = {0, 1}, as well as
    logistic regression, y = [0, 1].
    If you add `constant_to_optimal_zero` to the loss, you get half the
    Bernoulli/binomial deviance.

    More details: Inserting the predicted probability y_pred = expit(raw_prediction)
    in the loss gives the well known::

        loss(x_i) = - y_true_i * log(y_pred_i) - (1 - y_true_i) * log(1 - y_pred_i)
    c                 ót   •— t         ‰| �  t        «       t        «       d||¬«       t	        dddd«      | _        y ©Nr9   ©r    r!   r"   r#   r$   r   r:   T)r…   r-   r   r   r   r)   r†   s       €r,   r-   zHalfBinomialLoss.__init__Ã  s>   ø€ Ü‰ÑÜ$Ó&Ü“ØØØð 	ô 	
ô  (¨¨1¨d°DÓ9ˆÕr.   c                 óR   — t        ||«      t        d|z
  d|z
  «      z   }|�||z  }|S ry   r   r³   s       r,   ri   z)HalfBinomialLoss.constant_to_optimal_zeroÍ  s7   € ä�V˜VÓ$¤u¨Q°©Z¸¸V¹Ó'DÑDˆØÐ$Ø�MÑ!ˆDØˆr.   c                 ó4  — |j                   dk(  r#|j                  d   dk(  r|j                  d«      }t        j                  |j                  d   df|j
                  ¬«      }| j                  j                  |«      |dd…df<   d|dd…df   z
  |dd…df<   |S ©a=  Predict probabilities.

        Parameters
        ----------
        raw_prediction : array of shape (n_samples,) or (n_samples, 1)
            Raw prediction values (in link space).

        Returns
        -------
        proba : array of shape (n_samples, 2)
            Element-wise class probabilities.
        r9   r:   r   rH   N©rB   rC   rD   r'   rq   rI   r!   Úinverse©r+   r=   Úprobas      r,   Úpredict_probazHalfBinomialLoss.predict_probaÔ  ó”   € ð ×Ñ !Ò#¨×(<Ñ(<¸QÑ(?À1Ò(DØ+×3Ñ3°AÓ6ˆNÜ—‘˜.×.Ñ.¨qÑ1°1Ð5¸^×=QÑ=QÔRˆØ—i‘i×'Ñ'¨Ó7ˆŠa�ˆd‰Ø˜%¢ 1 ™+‘oˆŠa�ˆd‰Øˆr.   rv   rz   ©r{   r|   r}   r~   r-   ri   rÓ   r‰   rŠ   s   @r,   rÈ   rÈ   £  s   ø„ ñõ>:óör.   rÈ   c                   óL   ‡ — e Zd ZdZdZdˆ fd„	Zd„ Zd	d„Zd„ Z	 	 	 	 d
d„Z	ˆ xZ
S )ÚHalfMultinomialLossa/  Categorical cross-entropy loss, for multiclass classification.

    Domain:
    y_true in {0, 1, 2, 3, .., n_classes - 1}
    y_pred has n_classes elements, each element in (0, 1)

    Link:
    y_pred = softmax(raw_prediction)

    Note: We assume y_true to be already label encoded. The inverse link is
    softmax. But the full link function is the symmetric multinomial logit
    function.

    For a given sample x_i, the categorical cross-entropy loss is defined as
    the negative log-likelihood of the multinomial distribution, it
    generalizes the binary cross-entropy to more than 2 classes::

        loss_i = log(sum(exp(raw_pred_{i, k}), k=0..n_classes-1))
                - sum(y_true_{i, k} * raw_pred_{i, k}, k=0..n_classes-1)

    See [1].

    Note that for the hessian, we calculate only the diagonal part in the
    classes: If the full hessian for classes k and l and sample i is H_i_k_l,
    we calculate H_i_k_k, i.e. k=l.

    Parameters
    ----------
    sample_weight : {None, ndarray}
        If sample_weight is None, the hessian might be constant.

    n_classes : {None, int}
        The number of classes for classification, else None.

    xp : module or None
        Array namespace module. Ignored by the Cython implementation.

    device : device or None
        A device object. Ignored by the Cython implementation.

    References
    ----------
    .. [1] :arxiv:`Simon, Noah, J. Friedman and T. Hastie.
        "A Blockwise Descent Algorithm for Group-penalized Multiresponse and
        Multinomial Regression".
        <1311.6529>`
    Tc                 óà   •— t         ‰| �  t        «       t        «       |||¬«       t	        dt
        j                  dd«      | _        t	        dddd«      | _        d | _	        d | _
        d | _        y )NrË   r   TFr:   )r…   r-   r	   r   r   r'   r(   r)   r*   Úclass_indexing_offsetsÚ
y_true_intÚy_true_one_hot©r+   r>   r"   r#   r$   r‡   s        €r,   r-   zHalfMultinomialLoss.__init__  so   ø€ Ü‰ÑÜ'Ó)Ü!Ó#ØØØð 	ô 	
ô  (¨¬2¯6©6°4¸Ó?ˆÔÜ'¨¨1¨e°UÓ;ˆÔð '+ˆÔ#ØˆŒØ"ˆÕr.   c                 ó’   — | j                   j                  |«      xr+ t        j                  |j	                  t
        «      |k(  «      S r0   )r)   r1   r'   ÚallÚastypeÚintr2   s     r,   r4   z#HalfMultinomialLoss.in_y_true_range.  s6   € ð ×#Ñ#×,Ñ,¨QÓ/ÒN´B·F±F¸1¿8¹8ÄC»=ÈAÑ;MÓ4NÐNr.   c                 ó¼  — t        j                  | j                  |j                  ¬«      }t        j                  |j                  «      j
                  }t        | j                  «      D ]@  }t        j                  ||k(  |d¬«      ||<   t        j                  ||   |d|z
  «      ||<   ŒB | j                  j                  |ddd…f   «      j                  d«      S )a-  Compute raw_prediction of an intercept-only model.

        This is the softmax of the weighted average of the target, i.e. over
        the samples axis=0.

        Parameters
        ----------
        y_true : array-like of shape (n_samples,)
            Observed, true target values.

        sample_weight : None or array of shape (n_samples,), default=None
            Sample weights.

        Returns
        -------
        raw_prediction : numpy scalar or array of shape (n_classes,)
            Raw predictions of an intercept-only model.
        rH   r   rX   r:   Néÿÿÿÿ)r'   Úzerosr"   rI   r[   r\   ÚrangerU   ra   r!   Úreshape)r+   r<   r>   Úoutr\   Úks         r,   re   z&HalfMultinomialLoss.fit_intercept_only7  s¬   € ô& �h‰h�t—~‘~¨V¯\©\Ô:ˆÜ�h‰h�v—|‘|Ó$×(Ñ(ˆÜ�t—~‘~Ó&ò 	3ˆAÜ—Z‘Z ¨!¡°]ÈÔKˆC�‰FÜ—W‘W˜S ™V S¨!¨c©'Ó2ˆC�ŠFð	3ð �y‰y�~‰~˜c $ª '™lÓ+×3Ñ3°BÓ7Ð7r.   c                 ó8   — | j                   j                  |«      S )a=  Predict probabilities.

        Parameters
        ----------
        raw_prediction : array of shape (n_samples, n_classes)
            Raw prediction values (in link space).

        Returns
        -------
        proba : array of shape (n_samples, n_classes)
            Element-wise class probabilities.
        )r!   rÐ   )r+   r=   s     r,   rÓ   z!HalfMultinomialLoss.predict_probaQ  s   € ð �y‰y× Ñ  Ó0Ð0r.   c                 ó  — |€C|€+t        j                  |«      }t        j                  |«      }n-t        j                  |«      }n|€t        j                  |«      }| j                  j                  ||||||¬«       ||fS )aK  Compute gradient and class probabilities fow raw_prediction.

        Parameters
        ----------
        y_true : C-contiguous array of shape (n_samples,)
            Observed, true target values.
        raw_prediction : array of shape (n_samples, n_classes)
            Raw prediction values (in link space).
        sample_weight : None or C-contiguous array of shape (n_samples,)
            Sample weights.
        gradient_out : None or array of shape (n_samples, n_classes)
            A location into which the gradient is stored. If None, a new array
            might be created.
        proba_out : None or array of shape (n_samples, n_classes)
            A location into which the class probabilities are stored. If None,
            a new array might be created.
        n_threads : int, default=1
            Might use openmp thread parallelism.

        Returns
        -------
        gradient : array of shape (n_samples, n_classes)
            Element-wise gradients.

        proba : array of shape (n_samples, n_classes)
            Element-wise class probabilities.
        )r<   r=   r>   rJ   Ú	proba_outr@   )r'   rA   r    Úgradient_proba)r+   r<   r=   r>   rJ   rê   r@   s          r,   rë   z"HalfMultinomialLoss.gradient_proba`  s…   € ðH ÐØÐ Ü!Ÿ}™}¨^Ó<�ÜŸM™M¨.Ó9‘	ä!Ÿ}™}¨YÓ7‘ØÐÜŸ™ lÓ3ˆIà�
‰
×!Ñ!ØØ)Ø'Ø%ØØð 	"ô 	
ð ˜YÐ&Ð&r.   ©Né   NNrz   rx   )r{   r|   r}   r~   rp   r-   r4   re   rÓ   rë   r‰   rŠ   s   @r,   r×   r×   ê  s8   ø„ ñ.ð` €Mõ#ò"Oó8ò41ð& ØØØ÷5'r.   r×   c                   ó2   ‡ — e Zd ZdZdˆ fd„	Zdd„Zd„ Zˆ xZS )ÚExponentialLossa"  Exponential loss with (half) logit link, for binary classification.

    This is also know as boosting loss.

    Domain:
    y_true in [0, 1], i.e. regression on the unit interval
    y_pred in (0, 1), i.e. boundaries excluded

    Link:
    y_pred = expit(2 * raw_prediction)

    For a given sample x_i, the exponential loss is defined as::

        loss(x_i) = y_true_i * exp(-raw_pred_i)) + (1 - y_true_i) * exp(raw_pred_i)

    See:
    - J. Friedman, T. Hastie, R. Tibshirani.
      "Additive logistic regression: a statistical view of boosting (With discussion
      and a rejoinder by the authors)." Ann. Statist. 28 (2) 337 - 407, April 2000.
      https://doi.org/10.1214/aos/1016218223
    - A. Buja, W. Stuetzle, Y. Shen. (2005).
      "Loss Functions for Binary Class Probability Estimation and Classification:
      Structure and Applications."

    Note that the formulation works for classification, y = {0, 1}, as well as
    "exponential logistic" regression, y = [0, 1].
    Note that this is a proper scoring rule, but without it's canonical link.

    More details: Inserting the predicted probability
    y_pred = expit(2 * raw_prediction) in the loss gives::

        loss(x_i) = y_true_i * sqrt((1 - y_pred_i) / y_pred_i)
            + (1 - y_true_i) * sqrt(y_pred_i / (1 - y_pred_i))
    c                 ót   •— t         ‰| �  t        «       t        «       d||¬«       t	        dddd«      | _        y rÊ   )r…   r-   r   r   r   r)   r†   s       €r,   r-   zExponentialLoss.__init__¼  s>   ø€ Ü‰ÑÜ#Ó%Ü“ØØØð 	ô 	
ô  (¨¨1¨d°DÓ9ˆÕr.   c                 óP   — dt        j                  |d|z
  z  «      z  }|�||z  }|S )Néþÿÿÿr:   )r'   Úsqrtr³   s       r,   ri   z(ExponentialLoss.constant_to_optimal_zeroÆ  s3   € à”B—G‘G˜F a¨&¡jÑ1Ó2Ñ2ˆØÐ$Ø�MÑ!ˆDØˆr.   c                 ó4  — |j                   dk(  r#|j                  d   dk(  r|j                  d«      }t        j                  |j                  d   df|j
                  ¬«      }| j                  j                  |«      |dd…df<   d|dd…df   z
  |dd…df<   |S rÎ   rÏ   rÑ   s      r,   rÓ   zExponentialLoss.predict_probaÍ  rÔ   r.   rv   rz   rÕ   rŠ   s   @r,   rï   rï   ˜  s   ø„ ñ!õF:óör.   rï   )
Úsquared_errorÚabsolute_errorÚpinball_lossÚ
huber_lossÚpoisson_lossÚ
gamma_lossÚtweedie_lossÚbinomial_lossÚmultinomial_lossÚexponential_lossc                   óH   — e Zd ZdZ	 	 dd„Z	 	 	 dd„Z	 	 	 	 d	d„Z	 	 	 dd„Zy)
ÚArrayAPILossMixina¤  Mixin for loss classes that are compatible with the array API.

    Currently this mixin redefines methods:
    - __call__(...)
    - loss(...)
    - loss_gradient(...)
    - gradient(...)

    such that they work according to the array API specification.
    It uses the attributes self.xp and self.device from BaseLoss and it assumes that
    methods self._compute_loss and self._compute_gradient are implemented.
    Nc                 ól   — | j                  ||d¬«      }t        t        ||| j                  ¬«      «      S )a˜  Compute the weighted average loss for the array API losses.

        Parameters
        ----------
        y_true : C-contiguous array of shape (n_samples,)
            Observed, true target values.
        raw_prediction : C-contiguous array of shape (n_samples,) or array of             shape (n_samples, n_classes)
            Raw prediction values (in link space).
        sample_weight : None or C-contiguous array of shape (n_samples,)
            Sample weights.
        n_threads : int, default=1
            Ignored by the array API implementation.

        Returns
        -------
        loss : float
            Mean or averaged loss function.
        N©r<   r=   r>   )rT   r#   )rE   r    r   r#   )r+   r<   r=   r>   r@   Úloss_xps         r,   rV   zArrayAPILossMixin.__call__ÿ  s8   € ð4 —)‘)Ø¨.Èð ó 
ˆô ”X˜g¨}ÀÇÁÔIÓJÐJr.   c                 ó*   — | j                  |||¬«      S )a  Compute the pointwise loss value for each input.

        Parameters
        ----------
        y_true : C-contiguous array of shape (n_samples,)
            Observed, true target values.
        raw_prediction : C-contiguous array of shape (n_samples,) or array of             shape (n_samples, n_classes)
            Raw prediction values (in link space).
        sample_weight : None or C-contiguous array of shape (n_samples,)
            Sample weights.
        loss_out : None or C-contiguous array of shape (n_samples,)
            Ignored by the array API implementation.
        n_threads : int, default=1
            Ignored by the array API implementation.

        Returns
        -------
        loss : array of shape (n_samples,)
            Element-wise loss function.
        r  )Ú_compute_lossrF   s         r,   rE   zArrayAPILossMixin.loss  s%   € ð: ×!Ñ!ØØ)Ø'ð "ó 
ð 	
r.   c                 óZ   — | j                  |||¬«      }| j                  |||¬«      }||fS )aG  Compute loss and gradient w.r.t. raw_prediction for each input.

        Parameters
        ----------
        y_true : C-contiguous array of shape (n_samples,)
            Observed, true target values.
        raw_prediction : C-contiguous array of shape (n_samples,) or array of             shape (n_samples, n_classes)
            Raw prediction values (in link space).
        sample_weight : None or C-contiguous array of shape (n_samples,)
            Sample weights.
        loss_out : None or C-contiguous array of shape (n_samples,)
            Ignored by the array API implementation.
        gradient_out : None or C-contiguous array of shape (n_samples,) or array             of shape (n_samples, n_classes)
            Ignored by the array API implementation.
        n_threads : int, default=1
            Ignored by the array API implementation.

        Returns
        -------
        loss : array of shape (n_samples,)
            Element-wise loss function.

        gradient : array of shape (n_samples,) or (n_samples, n_classes)
            Element-wise gradients.
        r  )r  Ú_compute_gradient)	r+   r<   r=   r>   r?   rJ   r@   rE   rM   s	            r,   rK   zArrayAPILossMixin.loss_gradientA  sO   € ðH ×!Ñ!ØØ)Ø'ð "ó 
ˆð
 ×)Ñ)ØØ)Ø'ð *ó 
ˆð
 �Xˆ~Ðr.   c                 ó*   — | j                  |||¬«      S )ax  Compute gradient of loss w.r.t raw_prediction for each input.

        Parameters
        ----------
        y_true : C-contiguous array of shape (n_samples,)
            Observed, true target values.
        raw_prediction : C-contiguous array of shape (n_samples,) or array of             shape (n_samples, n_classes)
            Raw prediction values (in link space).
        sample_weight : None or C-contiguous array of shape (n_samples,)
            Sample weights.
        gradient_out : None or C-contiguous array of shape (n_samples,) or array             of shape (n_samples, n_classes)
            Ignored by the array API implementation.
        n_threads : int, default=1
            Ignored by the array API implementation.

        Returns
        -------
        gradient : array of shape (n_samples,) or (n_samples, n_classes)
            Element-wise gradients.
        r  )r  rN   s         r,   rM   zArrayAPILossMixin.gradientq  s%   € ð< ×%Ñ%ØØ)Ø'ð &ó 
ð 	
r.   ry   rw   rx   )r{   r|   r}   r~   rV   rE   rK   rM   r€   r.   r,   r   r   ñ  sK   „ ñð" ØóKðF ØØó!
ðN ØØØó.ðh ØØô"
r.   r   c                 óN  — | j                   |j                  k(  rg d¢ng d¢}|j                  | |d   k  ||j                  | |d   k  |j                  |«      |j                  | |d   k  |j	                  d|z   «      |j                  | |d   k  | d|z  z   | «      «      «      «      S )an  Numerically stable version of log(1 + exp(x)) that is compatible with
    the array API.

    Parameters
    ----------
    raw_prediction : C-contiguous array of shape (n_samples,) or array of         shape (n_samples, n_classes)
        Raw prediction values (in link space).
    raw_prediction_exp : C-contiguous array of shape (n_samples,) or array of         shape (n_samples, n_classes)
        Exponential of the raw prediction values.
    xp : module, default=None
        Array namespace module.

    Returns
    -------
    log1pexp : float
        Numerically stable value for log(1 + exp(raw_prediction)).
    )éÛÿÿÿrò   é   gfffff¦@@)éïÿÿÿrâ   é	   g333333-@r   r:   r9   g      ð?rí   )rI   rn   ÚwhereÚlog1pr¹   )r=   Úraw_prediction_expr#   Ú	constantss       r,   Ú	_log1pexpr  –  s¾   € ðj ×Ñ 2§:¡:Ò-ó 	âð ð
 �8‰8Ø˜) A™,Ñ&ØØ
�‰Ø˜i¨™lÑ*Ø�H‰HÐ'Ó(Ø�H‰HØ )¨A¡,Ñ.Ø—‘�sÐ/Ñ/Ó0Ø—‘Ø" i°¡lÑ2Ø" QÐ);Ñ%;Ñ;Ø"óóó	
óð r.   c                   ó8   — e Zd ZdZ	 	 	 	 dd„Z	 	 dd„Z	 	 dd„Zy)ÚHalfBinomialLossArrayAPIzHA version of the HalfBinomialLoss that is compatible with the array API.Nc                 ó”   — | j                   j                  |«      }| j                  ||||¬«      }| j                  ||||¬«      }	||	fS ©N)r<   r=   r>   r  ©r#   Úexpr  r  ©
r+   r<   r=   r>   r?   rJ   r@   r  rE   rM   s
             r,   rK   z&HalfBinomialLossArrayAPI.loss_gradientä  óg   € ð "ŸW™WŸ[™[¨Ó8ÐØ×!Ñ!ØØ)Ø'Ø1ð	 "ó 
ˆð ×)Ñ)ØØ)Ø'Ø1ð	 *ó 
ˆð �Xˆ~Ðr.   c                 óŽ   — |€| j                   j                  |«      }t        ||| j                   ¬«      }|||z  z
  }|�||z  }|S )N)r=   r  r#   )r#   r  r  )r+   r<   r=   r>   r  Úlog1pexprE   s          r,   r  z&HalfBinomialLossArrayAPI._compute_lossü  sZ   € ð Ð%Ø!%§¡§¡¨^Ó!<ÐÜØ)Ø1Ø�w‰wô
ˆð
 ˜& >Ñ1Ñ1ˆØÐ$Ø�MÑ!ˆDØˆr.   c                 óâ   — | j                   }|€|j                  |«      }d|z  }|j                  ||j                  |j                  k(  rdndkD  d|z
  ||z  z
  d|z   z  ||z
  «      }|�||z  }|S )Nr:   r
  r  )r#   r  r  rI   rn   )r+   r<   r=   r>   r  r#   Úneg_raw_prediction_expÚgrads           r,   r  z*HalfBinomialLossArrayAPI._compute_gradient  s–   € ð �W‰WˆØÐ%Ø!#§¡¨Ó!7ÐØ!"Ð%7Ñ!7ÐØ�x‰xØ ^×%9Ñ%9¸R¿Z¹ZÒ%G™cÈSÑQØ�&‰j˜FÐ%;Ñ;Ñ;ØÐ)Ñ)ñ+à Ñ'ó	
ˆð Ð$Ø�MÑ!ˆDØˆr.   rx   ©NN©r{   r|   r}   r~   rK   r  r  r€   r.   r,   r  r  á  s2   „ ÙRð ØØØóð8 Øóð. Øôr.   r  c                   ó8   ‡ — e Zd ZdZdˆ fd„	Z	 dd„Z	 dd„Zˆ xZS )ÚHalfMultinomialLossArrayAPIa�  A version of the HalfMultinomialLoss that is compatible with the array API.

    Parameters
    ----------
    sample_weight : {None, ndarray}
        If sample_weight is None, the hessian might be constant.

    n_classes : {None, int}
        The number of classes for classification, else None.

    xp : module or None
        Array namespace module.

    device : device or None
        A device object.
    c                 óT   •— t         ‰| �  |||¬«       d | _        d | _        d | _        y )N)r"   r#   r$   )r…   r-   rÙ   rÚ   rÛ   rÜ   s        €r,   r-   z$HalfMultinomialLossArrayAPI.__init__7  s2   ø€ Ü‰Ñ 9°¸FÐÔCð '+ˆÔ#ØˆŒð #ˆÕr.   c                 ó¨  — | j                   }| j                  }t        |d|¬«      }| j                  €#|j	                  ||j
                  |¬«      | _        | j                  €2|j                  |j                  d   |¬«      | j                  z  | _        |j                  t        |«      | j                  | j                  z   «      }||z
  }|�||z  }|S )Nr:   )rY   r#   ©rI   r$   r   )r$   )r#   r$   r   rÚ   ÚasarrayÚint64rÙ   ÚarangerC   r"   Útaker   )	r+   r<   r=   r>   r#   r$   Úlog_sum_expÚtrue_label_probsrE   s	            r,   r  z)HalfMultinomialLossArrayAPI._compute_lossD  sÉ   € ð �W‰WˆØ—‘ˆÜ  °a¸BÔ?ˆØ�?‰?Ð"Ø Ÿj™j¨°r·x±xÈ˜jÓOˆDŒOà×&Ñ&Ð.à—	‘	˜&Ÿ,™, q™/°&�	Ó9¸D¿N¹NÑJð Ô'ð Ÿ7™7Ü�>Ó" D§O¡O°d×6QÑ6QÑ$Qó
Ðð Ð-Ñ-ˆØÐ$Ø�MÑ!ˆDØˆr.   c                 ó^  — | j                   }| j                  }| j                  €`| j                  €#|j	                  ||j
                  |¬«      | _        t        | j                  | j                  |j                  ¬«      | _        t        |«      }|| j                  z  }|�||d d …d f   z  }|S )Nr&  )Únum_classesrI   )
r#   r$   rÛ   rÚ   r'  r(  r   r"   rI   r   )r+   r<   r=   r>   r#   Údevice_r  s          r,   r  z-HalfMultinomialLossArrayAPI._compute_gradient\  s¦   € ð �W‰WˆØ—+‘+ˆØ×ÑÐ&Ø�‰Ð&Ø"$§*¡*¨V¸2¿8¹8ÈG *Ó"T�”ä")Ø—‘Ø ŸN™NØ$×*Ñ*ô#ˆDÔô
 �~Ó&ˆð 	�×#Ñ#Ñ#ˆØÐ$Ø�M¢! T 'Ñ*Ñ*ˆDØˆr.   rì   rz   )r{   r|   r}   r~   r-   r  r  r‰   rŠ   s   @r,   r#  r#  %  s!   ø„ ñõ"#ð" ó	ð8 ÷	r.   r#  c                   ó8   — e Zd ZdZ	 	 	 	 dd„Z	 	 dd„Z	 	 dd„Zy)ÚHalfPoissonLossArrayAPIzGA version of the HalfPoissonLoss that is compatible with the array API.Nc                 ó”   — | j                   j                  |«      }| j                  ||||¬«      }| j                  ||||¬«      }	||	fS r  r  r  s
             r,   rK   z%HalfPoissonLossArrayAPI.loss_gradient€  r  r.   c                 ó^   — |€| j                   j                  |«      }|||z  z
  }|�||z  }|S rz   ©r#   r  )r+   r<   r=   r>   r  rE   s         r,   r  z%HalfPoissonLossArrayAPI._compute_loss˜  sA   € ð Ð%Ø!%§¡§¡¨^Ó!<ÐØ! F¨^Ñ$;Ñ;ˆØÐ$Ø�MÑ!ˆDØˆr.   c                 óX   — |€| j                   j                  |«      }||z
  }|�||z  }|S rz   r4  )r+   r<   r=   r>   r  r  s         r,   r  z)HalfPoissonLossArrayAPI._compute_gradient¦  s<   € ð Ð%Ø!%§¡§¡¨^Ó!<ÐØ! FÑ*ˆØÐ$Ø�MÑ!ˆDØˆr.   rx   r   r!  r€   r.   r,   r1  r1  }  s2   „ ÙQð ØØØóð8 Øóð$ Øôr.   r1  )7r~   rž   Únumpyr'   Úscipy.specialr   Úsklearn._loss._lossr   r   r   r   r	   r
   r   r   r   r   r   Úsklearn._loss.linkr   r   r   r   r   r   Ú!sklearn.externals.array_api_extrar   Úsklearn.utilsr   Úsklearn.utils._array_apir   r   r   Úsklearn.utils.extmathr   Úsklearn.utils.statsr   r   r‚   rŒ   r•   r¦   r°   r¶   r»   rÅ   rÈ   r×   rï   Ú_LOSSESr   r  r  r#  r1  r€   r.   r,   ú<module>r@     se  ðñó* ã Ý ÷÷ ÷ ñ ÷÷ õ 6Ý &÷ñ õ
 *Ý 4÷*T!ñ T!ôn6�xô 6ô2$C�Hô $CôN=�(ô =ô@I@�ô I@ôX�hô ôD�Hô ô>?�hô ?ôD-E˜hô -Eô`D�xô DôNk'˜(ô k'ô\H�hô HðX &Ø#ØØØ#ØØ#Ø%Ø+Ø'ñ€÷b
ñ b
òJHôVAÐ0Ð2Bô AôHUÐ"3Ð5Hô Uôp5Ð/°õ 5r.   