ó
    §ñ:i¡  ã                   ó°  • S r SSKrSSKrSSKJr  SSKJr  SSKJ	r	  SSK
JrJrJrJrJrJrJrJrJrJrJr  SS	KJrJrJrJrJrJr   " S
 S5      r " S S\5      r " S S\5      r " S S\5      r  " S S\5      r! " S S\5      r" " S S\5      r# " S S\5      r$ " S S\5      r% " S S\5      r& " S S\5      r' " S  S!\5      r(\\\ \!\"\#\$\&\'\(S".
r)g)#z¹
This module contains loss classes suitable for fitting.

It is not part of the public API.
Specific losses are used for regression, binary classification or multiclass
classification.
é    N©Úxlogyé   )Úcheck_scalar)Ú_weighted_percentileé   )ÚCyAbsoluteErrorÚCyExponentialLossÚCyHalfBinomialLossÚCyHalfGammaLossÚCyHalfMultinomialLossÚCyHalfPoissonLossÚCyHalfSquaredErrorÚCyHalfTweedieLossÚCyHalfTweedieLossIdentityÚCyHuberLossÚCyPinballLoss)ÚHalfLogitLinkÚIdentityLinkÚIntervalÚ	LogitLinkÚLogLinkÚMultinomialLogitc                   ó¾   • \ rS rSrSrSrSrSrSS jrS r	S r
   SS	 jr    SS
 jr   SS jr    SS jrSS jrSS jrSS jr\R&                  S4S jrSrg)ÚBaseLosséC   a|  Base class for a loss function of 1-dimensional targets.

Conventions:

    - y_true.shape = sample_weight.shape = (n_samples,)
    - y_pred.shape = raw_prediction.shape = (n_samples,)
    - If is_multiclass is true (multiclass classification), then
      y_pred.shape = raw_prediction.shape = (n_samples, n_classes)
      Note that this corresponds to the return value of decision_function.

y_true, y_pred, sample_weight and raw_prediction must either be all float64
or all float32.
gradient and hessian must be either both float64 or both float32.

Note that y_pred = link.inverse(raw_prediction).

Specific loss classes can inherit specific link classes to satisfy
BaseLink's abstractmethods.

Parameters
----------
sample_weight : {None, ndarray}
    If sample_weight is None, the hessian might be constant.
n_classes : {None, int}
    The number of classes for classification, else None.

Attributes
----------
closs: CyLossFunction
link : BaseLink
interval_y_true : Interval
    Valid interval for y_true
interval_y_pred : Interval
    Valid Interval for y_pred
differentiable : bool
    Indicates whether or not loss function is differentiable in
    raw_prediction everywhere.
need_update_leaves_values : bool
    Indicates whether decision trees in gradient boosting need to uptade
    leave values after having been fit to the (negative) gradients.
approx_hessian : bool
    Indicates whether the hessian is approximated or exact. If,
    approximated, it should be larger or equal to the exact one.
constant_hessian : bool
    Indicates whether the hessian is one for this loss.
is_multiclass : bool
    Indicates whether n_classes > 2 is allowed.
TFNc                 óÚ   • Xl         X l        SU l        SU l        X0l        [        [        R                  * [        R                  SS5      U l        U R                  R                  U l	        g )NF)
ÚclossÚlinkÚapprox_hessianÚconstant_hessianÚ	n_classesr   ÚnpÚinfÚinterval_y_trueÚinterval_y_pred)Úselfr   r   r"   s       ÚU/srv/projetos/modelo_ml_acdoc/venv/lib/python3.13/site-packages/sklearn/_loss/loss.pyÚ__init__ÚBaseLoss.__init__‚   sP   € ØŒ
ØŒ	Ø#ˆÔØ %ˆÔØ"ŒÜ'¬¯©¨´·±¸ÀÓFˆÔØ#Ÿy™y×8Ñ8ˆÕó    c                 ó8   • U R                   R                  U5      $ ©zUReturn True if y is in the valid range of y_true.

Parameters
----------
y : ndarray
)r%   Úincludes©r'   Úys     r(   Úin_y_true_rangeÚBaseLoss.in_y_true_range‹   ó   € ð ×#Ñ#×,Ñ,¨QÓ/Ð/r+   c                 ó8   • U R                   R                  U5      $ )zUReturn True if y is in the valid range of y_pred.

Parameters
----------
y : ndarray
)r&   r.   r/   s     r(   Úin_y_pred_rangeÚBaseLoss.in_y_pred_range”   r3   r+   c                 óÚ   • Uc  [         R                  " U5      nUR                  S:X  a$  UR                  S   S:X  a  UR	                  S5      nU R
                  R                  UUUUUS9  U$ )aº  Compute the pointwise loss value for each input.

Parameters
----------
y_true : C-contiguous array of shape (n_samples,)
    Observed, true target values.
raw_prediction : C-contiguous array of shape (n_samples,) or array of             shape (n_samples, n_classes)
    Raw prediction values (in link space).
sample_weight : None or C-contiguous array of shape (n_samples,)
    Sample weights.
loss_out : None or C-contiguous array of shape (n_samples,)
    A location into which the result is stored. If None, a new array
    might be created.
n_threads : int, default=1
    Might use openmp thread parallelism.

Returns
-------
loss : array of shape (n_samples,)
    Element-wise loss function.
r   r   ©Úy_trueÚraw_predictionÚsample_weightÚloss_outÚ	n_threads)r#   Ú
empty_likeÚndimÚshapeÚsqueezer   Úloss)r'   r9   r:   r;   r<   r=   s         r(   rB   ÚBaseLoss.loss�   ss   € ð< ÑÜ—}’} VÓ,ˆHà×Ñ !Ó#¨×(<Ñ(<¸QÑ(?À1Ó(DØ+×3Ñ3°AÓ6ˆNà�
‰
�‰ØØ)Ø'ØØð 	ñ 	
ð ˆr+   c           	      óú  • UcO  Uc-  [         R                  " U5      n[         R                  " U5      nO@[         R                  " XR                  S9nO!Uc  [         R                  " X$R                  S9nUR                  S:X  a$  UR                  S   S:X  a  UR                  S5      nUR                  S:X  a$  UR                  S   S:X  a  UR                  S5      nU R                  R                  UUUUUUS9  XE4$ )a÷  Compute loss and gradient w.r.t. raw_prediction for each input.

Parameters
----------
y_true : C-contiguous array of shape (n_samples,)
    Observed, true target values.
raw_prediction : C-contiguous array of shape (n_samples,) or array of             shape (n_samples, n_classes)
    Raw prediction values (in link space).
sample_weight : None or C-contiguous array of shape (n_samples,)
    Sample weights.
loss_out : None or C-contiguous array of shape (n_samples,)
    A location into which the loss is stored. If None, a new array
    might be created.
gradient_out : None or C-contiguous array of shape (n_samples,) or array             of shape (n_samples, n_classes)
    A location into which the gradient is stored. If None, a new array
    might be created.
n_threads : int, default=1
    Might use openmp thread parallelism.

Returns
-------
loss : array of shape (n_samples,)
    Element-wise loss function.

gradient : array of shape (n_samples,) or (n_samples, n_classes)
    Element-wise gradients.
©Údtyper   r   )r9   r:   r;   r<   Úgradient_outr=   )r#   r>   rF   r?   r@   rA   r   Úloss_gradient)r'   r9   r:   r;   r<   rG   r=   s          r(   rH   ÚBaseLoss.loss_gradientÊ   sï   € ðL ÑØÑ#ÜŸ=š=¨Ó0�Ü!Ÿ}š}¨^Ó<‘äŸ=š=¨×7IÑ7IÑJ‘ØÑ!ÜŸ=š=¨¿~¹~ÑNˆLð ×Ñ !Ó#¨×(<Ñ(<¸QÑ(?À1Ó(DØ+×3Ñ3°AÓ6ˆNØ×Ñ Ó! l×&8Ñ&8¸Ñ&;¸qÓ&@Ø'×/Ñ/°Ó2ˆLà�
‰
× Ñ ØØ)Ø'ØØ%Øð 	!ñ 	
ð Ð%Ð%r+   c                 óB  • Uc  [         R                  " U5      nUR                  S:X  a$  UR                  S   S:X  a  UR	                  S5      nUR                  S:X  a$  UR                  S   S:X  a  UR	                  S5      nU R
                  R                  UUUUUS9  U$ )a  Compute gradient of loss w.r.t raw_prediction for each input.

Parameters
----------
y_true : C-contiguous array of shape (n_samples,)
    Observed, true target values.
raw_prediction : C-contiguous array of shape (n_samples,) or array of             shape (n_samples, n_classes)
    Raw prediction values (in link space).
sample_weight : None or C-contiguous array of shape (n_samples,)
    Sample weights.
gradient_out : None or C-contiguous array of shape (n_samples,) or array             of shape (n_samples, n_classes)
    A location into which the result is stored. If None, a new array
    might be created.
n_threads : int, default=1
    Might use openmp thread parallelism.

Returns
-------
gradient : array of shape (n_samples,) or (n_samples, n_classes)
    Element-wise gradients.
r   r   )r9   r:   r;   rG   r=   )r#   r>   r?   r@   rA   r   Úgradient)r'   r9   r:   r;   rG   r=   s         r(   rK   ÚBaseLoss.gradient	  s¨   € ð> ÑÜŸ=š=¨Ó8ˆLð ×Ñ !Ó#¨×(<Ñ(<¸QÑ(?À1Ó(DØ+×3Ñ3°AÓ6ˆNØ×Ñ Ó! l×&8Ñ&8¸Ñ&;¸qÓ&@Ø'×/Ñ/°Ó2ˆLà�
‰
×ÑØØ)Ø'Ø%Øð 	ñ 	
ð Ðr+   c           	      óB  • UcG  Uc-  [         R                  " U5      n[         R                  " U5      nO0[         R                  " U5      nOUc  [         R                  " U5      nUR                  S:X  a$  UR                  S   S:X  a  UR	                  S5      nUR                  S:X  a$  UR                  S   S:X  a  UR	                  S5      nUR                  S:X  a$  UR                  S   S:X  a  UR	                  S5      nU R
                  R                  UUUUUUS9  XE4$ )aG  Compute gradient and hessian of loss w.r.t raw_prediction.

Parameters
----------
y_true : C-contiguous array of shape (n_samples,)
    Observed, true target values.
raw_prediction : C-contiguous array of shape (n_samples,) or array of             shape (n_samples, n_classes)
    Raw prediction values (in link space).
sample_weight : None or C-contiguous array of shape (n_samples,)
    Sample weights.
gradient_out : None or C-contiguous array of shape (n_samples,) or array             of shape (n_samples, n_classes)
    A location into which the gradient is stored. If None, a new array
    might be created.
hessian_out : None or C-contiguous array of shape (n_samples,) or array             of shape (n_samples, n_classes)
    A location into which the hessian is stored. If None, a new array
    might be created.
n_threads : int, default=1
    Might use openmp thread parallelism.

Returns
-------
gradient : arrays of shape (n_samples,) or (n_samples, n_classes)
    Element-wise gradients.

hessian : arrays of shape (n_samples,) or (n_samples, n_classes)
    Element-wise hessians.
r   r   )r9   r:   r;   rG   Úhessian_outr=   )r#   r>   r?   r@   rA   r   Úgradient_hessian)r'   r9   r:   r;   rG   rN   r=   s          r(   rO   ÚBaseLoss.gradient_hessian:  s  € ðN ÑØÑ"Ü!Ÿ}š}¨^Ó<�Ü Ÿmšm¨NÓ;‘ä!Ÿ}š}¨[Ó9‘ØÑ ÜŸ-š-¨Ó5ˆKð ×Ñ !Ó#¨×(<Ñ(<¸QÑ(?À1Ó(DØ+×3Ñ3°AÓ6ˆNØ×Ñ Ó! l×&8Ñ&8¸Ñ&;¸qÓ&@Ø'×/Ñ/°Ó2ˆLØ×Ñ˜qÓ  [×%6Ñ%6°qÑ%9¸QÓ%>Ø%×-Ñ-¨aÓ0ˆKà�
‰
×#Ñ#ØØ)Ø'Ø%Ø#Øð 	$ñ 	
ð Ð(Ð(r+   c           
      óN   • [         R                  " U R                  UUSSUS9US9$ )a  Compute the weighted average loss.

Parameters
----------
y_true : C-contiguous array of shape (n_samples,)
    Observed, true target values.
raw_prediction : C-contiguous array of shape (n_samples,) or array of             shape (n_samples, n_classes)
    Raw prediction values (in link space).
sample_weight : None or C-contiguous array of shape (n_samples,)
    Sample weights.
n_threads : int, default=1
    Might use openmp thread parallelism.

Returns
-------
loss : float
    Mean or averaged loss function.
Nr8   ©Úweights)r#   ÚaveragerB   )r'   r9   r:   r;   r=   s        r(   Ú__call__ÚBaseLoss.__call__|  s:   € ô( �zŠzØ�I‰IØØ-Ø"ØØ#ð ð ð "ñ	
ð 		
r+   c                 ó  • [         R                  " XSS9nS[         R                  " UR                  5      R                  -  nU R
                  R                  [         R                  * :X  a  SnOKU R
                  R                  (       a  U R
                  R                  nOU R
                  R                  U-   nU R
                  R                  [         R                  :X  a  SnOKU R
                  R                  (       a  U R
                  R                  nOU R
                  R                  U-
  nUc  Uc  U R                  R                  U5      $ U R                  R                  [         R                  " X5U5      5      $ )a»  Compute raw_prediction of an intercept-only model.

This can be used as initial estimates of predictions, i.e. before the
first iteration in fit.

Parameters
----------
y_true : array-like of shape (n_samples,)
    Observed, true target values.
sample_weight : None or array of shape (n_samples,)
    Sample weights.

Returns
-------
raw_prediction : numpy scalar or array of shape (n_classes,)
    Raw predictions of an intercept-only model.
r   ©rS   Úaxisé
   N)r#   rT   ÚfinforF   Úepsr&   Úlowr$   Úlow_inclusiveÚhighÚhigh_inclusiver   Úclip)r'   r9   r;   Úy_predr\   Úa_minÚa_maxs          r(   Úfit_intercept_onlyÚBaseLoss.fit_intercept_only›  s  € ô( —’˜FÀÑBˆØ”2—8’8˜FŸL™LÓ)×-Ñ-Ñ-ˆà×Ñ×#Ñ#¬¯© wÓ.Ø‰EØ×!Ñ!×/×/Ø×(Ñ(×,Ñ,‰Eà×(Ñ(×,Ñ,¨sÑ2ˆEà×Ñ×$Ñ$¬¯©Ó.Ø‰EØ×!Ñ!×0×0Ø×(Ñ(×-Ñ-‰Eà×(Ñ(×-Ñ-°Ñ3ˆEà‰=˜U™]Ø—9‘9—>‘> &Ó)Ð)à—9‘9—>‘>¤"§'¢'¨&¸Ó"?Ó@Ð@r+   c                 ó.   • [         R                  " U5      $ )z`Calculate term dropped in loss.

With this term added, the loss of perfect predictions is zero.
)r#   Ú
zeros_like©r'   r9   r;   s      r(   Úconstant_to_optimal_zeroÚ!BaseLoss.constant_to_optimal_zeroÅ  s   € ô
 �}Š}˜VÓ$Ð$r+   ÚFc                 óX  • U[         R                  [         R                  4;  a  [        SU S35      eU R                  (       a  XR
                  4nOU4n[         R                  " XBUS9nU R                  (       a  [         R                  " SUS9nXV4$ [         R                  " XBUS9nXV4$ )aÅ  Initialize arrays for gradients and hessians.

Unless hessians are constant, arrays are initialized with undefined values.

Parameters
----------
n_samples : int
    The number of samples, usually passed to `fit()`.
dtype : {np.float64, np.float32}, default=np.float64
    The dtype of the arrays gradient and hessian.
order : {'C', 'F'}, default='F'
    Order of the arrays gradient and hessian. The default 'F' makes the arrays
    contiguous along samples.

Returns
-------
gradient : C-contiguous array of shape (n_samples,) or array of shape             (n_samples, n_classes)
    Empty array (allocated but not initialized) to be used as argument
    gradient_out.
hessian : C-contiguous array of shape (n_samples,), array of shape
    (n_samples, n_classes) or shape (1,)
    Empty (allocated but not initialized) array to be used as argument
    hessian_out.
    If constant_hessian is True (e.g. `HalfSquaredError`), the array is
    initialized to ``1``.
zCValid options for 'dtype' are np.float32 and np.float64. Got dtype=z	 instead.)r@   rF   Úorder)r   )r@   rF   )	r#   Úfloat32Úfloat64Ú
ValueErrorÚis_multiclassr"   Úemptyr!   Úones)r'   Ú	n_samplesrF   rn   r@   rK   Úhessians          r(   Úinit_gradient_and_hessianÚ"BaseLoss.init_gradient_and_hessianÌ  s¦   € ð8 œŸ™¤R§Z¡ZÐ0Ó0ÜðØ"˜G 9ð.óð ð
 ××Ø§¡Ð/‰Eà�LˆEÜ—8’8 %¸EÑBˆà× × ô
 —g’g D°Ñ6ˆGð Ð Ð ô —h’h U¸uÑEˆGàÐ Ð r+   )r    r   r!   r&   r%   r   r"   ©N)NNr   ©NNNr   ©Nr   )Ú__name__Ú
__module__Ú__qualname__Ú__firstlineno__Ú__doc__ÚdifferentiableÚneed_update_leaves_valuesrr   r)   r1   r5   rB   rH   rK   rO   rU   re   rj   r#   rp   rw   Ú__static_attributes__© r+   r(   r   r   C   s•   † ñ/ðt €NØ %ÐØ€Mô9ò0ò0ð ØØô+ðb ØØØô=&ðF ØØô/ðj ØØØô@)ôD
ô>(AôT%ð :<¿¹È3÷ 1!r+   r   c                   ó0   ^ • \ rS rSrSrSU 4S jjrSrU =r$ )ÚHalfSquaredErrori  a´  Half squared error with identity link, for regression.

Domain:
y_true and y_pred all real numbers

Link:
y_pred = raw_prediction

For a given sample x_i, half squared error is defined as::

    loss(x_i) = 0.5 * (y_true_i - raw_prediction_i)**2

The factor of 0.5 simplifies the computation of gradients and results in a
unit hessian (and is consistent with what is done in LightGBM). It is also
half the Normal distribution deviance.
c                 óT   >• [         TU ]  [        5       [        5       S9  US L U l        g )N©r   r   )Úsuperr)   r   r   r!   ©r'   r;   Ú	__class__s     €r(   r)   ÚHalfSquaredError.__init__  s(   ø€ Ü‰ÑÔ1Ó3¼,».ÐÑIØ -°Ð 5ˆÕr+   )r!   ry   ©r|   r}   r~   r   r€   r)   rƒ   Ú__classcell__©r‹   s   @r(   r†   r†     s   ø† ñ÷"6õ 6r+   r†   c                   óB   ^ • \ rS rSrSrSrSrSU 4S jjrSS jrSr	U =r
$ )	ÚAbsoluteErrori  a®  Absolute error with identity link, for regression.

Domain:
y_true and y_pred all real numbers

Link:
y_pred = raw_prediction

For a given sample x_i, the absolute error is defined as::

    loss(x_i) = |y_true_i - raw_prediction_i|

Note that the exact hessian = 0 almost everywhere (except at one point, therefore
differentiable = False). Optimization routines like in HGBT, however, need a
hessian > 0. Therefore, we assign 1.
FTc                 ób   >• [         TU ]  [        5       [        5       S9  SU l        US L U l        g )Nrˆ   T)r‰   r)   r	   r   r    r!   rŠ   s     €r(   r)   ÚAbsoluteError.__init__0  s/   ø€ Ü‰ÑœÓ0´|³~ÐÑFØ"ˆÔØ -°Ð 5ˆÕr+   c                 óJ   • Uc  [         R                  " USS9$ [        XS5      $ )ú}Compute raw_prediction of an intercept-only model.

This is the weighted median of the target, i.e. over the samples
axis=0.
r   ©rY   é2   )r#   Úmedianr   ri   s      r(   re   Ú AbsoluteError.fit_intercept_only5  s(   € ð Ñ Ü—9’9˜V¨!Ñ,Ð,ä'¨¸rÓBÐBr+   ©r    r!   ry   ©r|   r}   r~   r   r€   r�   r‚   r)   re   rƒ   rŽ   r�   s   @r(   r‘   r‘     s&   ø† ñð" €NØ $Ð÷6÷
	Cò 	Cr+   r‘   c                   óB   ^ • \ rS rSrSrSrSrSU 4S jjrS	S jrSr	U =r
$ )
ÚPinballLossiA  a3  Quantile loss aka pinball loss, for regression.

Domain:
y_true and y_pred all real numbers
quantile in (0, 1)

Link:
y_pred = raw_prediction

For a given sample x_i, the pinball loss is defined as::

    loss(x_i) = rho_{quantile}(y_true_i - raw_prediction_i)

    rho_{quantile}(u) = u * (quantile - 1_{u<0})
                      = -u *(1 - quantile)  if u < 0
                         u * quantile       if u >= 0

Note: 2 * PinballLoss(quantile=0.5) equals AbsoluteError().

Note that the exact hessian = 0 almost everywhere (except at one point, therefore
differentiable = False). Optimization routines like in HGBT, however, need a
hessian > 0. Therefore, we assign 1.

Additional Attributes
---------------------
quantile : float
    The quantile level of the quantile to be estimated. Must be in range (0, 1).
FTc           	      óª   >• [        US[        R                  SSSS9  [        TU ]  [        [        U5      S9[        5       S9  SU l        US L U l	        g )	NÚquantiler   r   Úneither©Útarget_typeÚmin_valÚmax_valÚinclude_boundaries)rŸ   rˆ   T)
r   ÚnumbersÚRealr‰   r)   r   Úfloatr   r    r!   )r'   r;   rŸ   r‹   s      €r(   r)   ÚPinballLoss.__init__b  s]   ø€ ÜØØÜŸ™ØØØ(ò	
ô 	‰ÑÜ¬¨x«Ñ9Ü“ð 	ñ 	
ð #ˆÔØ -°Ð 5ˆÕr+   c                 ó¨   • Uc-  [         R                  " USU R                  R                  -  SS9$ [	        XSU R                  R                  -  5      $ )r•   éd   r   r–   )r#   Ú
percentiler   rŸ   r   ri   s      r(   re   ÚPinballLoss.fit_intercept_onlyr  sM   € ð Ñ Ü—=’= ¨¨t¯z©z×/BÑ/BÑ)BÈÑKÐKä'Ø s¨T¯Z©Z×-@Ñ-@Ñ'@óð r+   rš   )Nç      à?ry   r›   r�   s   @r(   r�   r�   A  s$   ø† ñð: €NØ $Ð÷6÷ ò r+   r�   c                   óB   ^ • \ rS rSrSrSrSrSU 4S jjrS	S jrSr	U =r
$ )
Ú	HuberLossi€  aŽ  Huber loss, for regression.

Domain:
y_true and y_pred all real numbers
quantile in (0, 1)

Link:
y_pred = raw_prediction

For a given sample x_i, the Huber loss is defined as::

    loss(x_i) = 1/2 * abserr**2            if abserr <= delta
                delta * (abserr - delta/2) if abserr > delta

    abserr = |y_true_i - raw_prediction_i|
    delta = quantile(abserr, self.quantile)

Note: HuberLoss(quantile=1) equals HalfSquaredError and HuberLoss(quantile=0)
equals delta * (AbsoluteError() - delta/2).

Additional Attributes
---------------------
quantile : float
    The quantile level which defines the breaking point `delta` to distinguish
    between absolute error and squared error. Must be in range (0, 1).

 Reference
---------
.. [1] Friedman, J.H. (2001). :doi:`Greedy function approximation: A gradient
  boosting machine <10.1214/aos/1013203451>`.
  Annals of Statistics, 29, 1189-1232.
FTc           	      ó²   >• [        US[        R                  SSSS9  X l        [        TU ]  [        [        U5      S9[        5       S9  SU l	        S	U l
        g )
NrŸ   r   r   r    r¡   )Údeltarˆ   TF)r   r¦   r§   rŸ   r‰   r)   r   r¨   r   r    r!   )r'   r;   rŸ   r²   r‹   s       €r(   r)   ÚHuberLoss.__init__¥  s]   ø€ ÜØØÜŸ™ØØØ(ò	
ð !ŒÜ‰ÑÜ¤E¨%£LÑ1Ü“ð 	ñ 	
ð #ˆÔØ %ˆÕr+   c                 ó0  • Uc  [         R                  " USSS9nO[        XS5      nX-
  n[         R                  " U5      [         R                  " U R
                  R                  [         R                  " U5      5      -  nU[         R                  " XRS9-   $ )r•   r—   r   r–   rR   )	r#   r¬   r   ÚsignÚminimumr   r²   ÚabsrT   )r'   r9   r;   r˜   ÚdiffÚterms         r(   re   ÚHuberLoss.fit_intercept_only¶  sr   € ð Ñ Ü—]’] 6¨2°AÑ6‰Fä)¨&ÀÓDˆFØ‰ˆÜ�wŠw�t‹}œrŸzšz¨$¯*©*×*:Ñ*:¼B¿FºFÀ4»LÓIÑIˆØœŸ
š
 4Ñ?Ñ?Ð?r+   )r    r!   rŸ   )NgÍÌÌÌÌÌì?r®   ry   r›   r�   s   @r(   r°   r°   €  s'   ø† ñðB €NØ $Ð÷&÷"@ò @r+   r°   c                   ó:   ^ • \ rS rSrSrSU 4S jjrSS jrSrU =r$ )ÚHalfPoissonLossiÉ  aO  Half Poisson deviance loss with log-link, for regression.

Domain:
y_true in non-negative real numbers
y_pred in positive real numbers

Link:
y_pred = exp(raw_prediction)

For a given sample x_i, half the Poisson deviance is defined as::

    loss(x_i) = y_true_i * log(y_true_i/exp(raw_prediction_i))
                - y_true_i + exp(raw_prediction_i)

Half the Poisson deviance is actually the negative log-likelihood up to
constant terms (not involving raw_prediction) and simplifies the
computation of the gradients.
We also skip the constant term `y_true_i * log(y_true_i) - y_true_i`.
c                 ó„   >• [         TU ]  [        5       [        5       S9  [	        S[
        R                  SS5      U l        g )Nrˆ   r   TF)r‰   r)   r   r   r   r#   r$   r%   rŠ   s     €r(   r)   ÚHalfPoissonLoss.__init__Þ  s2   ø€ Ü‰ÑÔ0Ó2¼»ÐÑCÜ'¨¬2¯6©6°4¸Ó?ˆÕr+   c                 ó0   • [        X5      U-
  nUb  X2-  nU$ ry   r   ©r'   r9   r;   r¹   s       r(   rj   Ú(HalfPoissonLoss.constant_to_optimal_zeroâ  s$   € Ü�VÓ$ vÑ-ˆØÑ$ØÑ!ˆDØˆr+   ©r%   ry   ©	r|   r}   r~   r   r€   r)   rj   rƒ   rŽ   r�   s   @r(   r¼   r¼   É  s   ø† ñ÷(@÷ò r+   r¼   c                   ó:   ^ • \ rS rSrSrSU 4S jjrSS jrSrU =r$ )ÚHalfGammaLossié  a&  Half Gamma deviance loss with log-link, for regression.

Domain:
y_true and y_pred in positive real numbers

Link:
y_pred = exp(raw_prediction)

For a given sample x_i, half Gamma deviance loss is defined as::

    loss(x_i) = log(exp(raw_prediction_i)/y_true_i)
                + y_true/exp(raw_prediction_i) - 1

Half the Gamma deviance is actually proportional to the negative log-
likelihood up to constant terms (not involving raw_prediction) and
simplifies the computation of the gradients.
We also skip the constant term `-log(y_true_i) - 1`.
c                 ó„   >• [         TU ]  [        5       [        5       S9  [	        S[
        R                  SS5      U l        g )Nrˆ   r   F)r‰   r)   r   r   r   r#   r$   r%   rŠ   s     €r(   r)   ÚHalfGammaLoss.__init__ý  s1   ø€ Ü‰ÑœÓ0´w³yÐÑAÜ'¨¬2¯6©6°5¸%Ó@ˆÕr+   c                 óH   • [         R                  " U5      * S-
  nUb  X2-  nU$ r{   )r#   ÚlogrÀ   s       r(   rj   Ú&HalfGammaLoss.constant_to_optimal_zero  s)   € Ü—’�v“ˆ Ñ"ˆØÑ$ØÑ!ˆDØˆr+   rÂ   ry   rÃ   r�   s   @r(   rÅ   rÅ   é  s   ø† ñ÷&A÷ò r+   rÅ   c                   ó:   ^ • \ rS rSrSrSU 4S jjrSS jrSrU =r$ )ÚHalfTweedieLossi  a«  Half Tweedie deviance loss with log-link, for regression.

Domain:
y_true in real numbers for power <= 0
y_true in non-negative real numbers for 0 < power < 2
y_true in positive real numbers for 2 <= power
y_pred in positive real numbers
power in real numbers

Link:
y_pred = exp(raw_prediction)

For a given sample x_i, half Tweedie deviance loss with p=power is defined
as::

    loss(x_i) = max(y_true_i, 0)**(2-p) / (1-p) / (2-p)
                - y_true_i * exp(raw_prediction_i)**(1-p) / (1-p)
                + exp(raw_prediction_i)**(2-p) / (2-p)

Taking the limits for p=0, 1, 2 gives HalfSquaredError with a log link,
HalfPoissonLoss and HalfGammaLoss.

We also skip constant terms, but those are different for p=0, 1, 2.
Therefore, the loss is not continuous in `power`.

Note furthermore that although no Tweedie distribution exists for
0 < power < 1, it still gives a strictly consistent scoring function for
the expectation.
c                 ó¢  >• [         TU ]  [        [        U5      S9[	        5       S9  U R
                  R                  S::  a1  [        [        R                  * [        R                  SS5      U l
        g U R
                  R                  S:  a"  [        S[        R                  SS5      U l
        g [        S[        R                  SS5      U l
        g ©N)Úpowerrˆ   r   Fr   T)r‰   r)   r   r¨   r   r   rÏ   r   r#   r$   r%   ©r'   r;   rÏ   r‹   s      €r(   r)   ÚHalfTweedieLoss.__init__'  s—   ø€ Ü‰ÑÜ#¬%°«,Ñ7Ü“ð 	ñ 	
ð �:‰:×Ñ˜qÓ Ü#+¬R¯V©V¨G´R·V±V¸UÀEÓ#JˆDÕ Ø�Z‰Z×Ñ Ó!Ü#+¨A¬r¯v©v°t¸UÓ#CˆDÕ ä#+¨A¬r¯v©v°u¸eÓ#DˆDÕ r+   c                 óÜ  • U R                   R                  S:X  a  [        5       R                  XS9$ U R                   R                  S:X  a  [	        5       R                  XS9$ U R                   R                  S:X  a  [        5       R                  XS9$ U R                   R                  n[        R                  " [        R                  " US5      SU-
  5      SU-
  -  SU-
  -  nUb  XB-  nU$ )Nr   )r9   r;   r   r   )r   rÏ   r†   rj   r¼   rÅ   r#   Úmaximum)r'   r9   r;   Úpr¹   s        r(   rj   Ú(HalfTweedieLoss.constant_to_optimal_zero3  sê   € Ø�:‰:×Ñ˜qÓ Ü#Ó%×>Ñ>Øð ?ð ð ð �Z‰Z×Ñ Ó"Ü"Ó$×=Ñ=Øð >ð ð ð �Z‰Z×Ñ Ó"Ü “?×;Ñ;Øð <ð ð ð —
‘
× Ñ ˆAÜ—8’8œBŸJšJ v¨qÓ1°1°q±5Ó9¸QÀ¹UÑCÀqÈ1ÁuÑMˆDØÑ(ØÑ%�ØˆKr+   rÂ   ©Ng      ø?ry   rÃ   r�   s   @r(   rÌ   rÌ     s   ø† ñ÷<
E÷ò r+   rÌ   c                   ó0   ^ • \ rS rSrSrSU 4S jjrSrU =r$ )ÚHalfTweedieLossIdentityiH  a"  Half Tweedie deviance loss with identity link, for regression.

Domain:
y_true in real numbers for power <= 0
y_true in non-negative real numbers for 0 < power < 2
y_true in positive real numbers for 2 <= power
y_pred in positive real numbers for power != 0
y_pred in real numbers for power = 0
power in real numbers

Link:
y_pred = raw_prediction

For a given sample x_i, half Tweedie deviance loss with p=power is defined
as::

    loss(x_i) = max(y_true_i, 0)**(2-p) / (1-p) / (2-p)
                - y_true_i * raw_prediction_i**(1-p) / (1-p)
                + raw_prediction_i**(2-p) / (2-p)

Note that the minimum value of this loss is 0.

Note furthermore that although no Tweedie distribution exists for
0 < power < 1, it still gives a strictly consistent scoring function for
the expectation.
c                 óz  >• [         TU ]  [        [        U5      S9[	        5       S9  U R
                  R                  S::  a1  [        [        R                  * [        R                  SS5      U l
        O]U R
                  R                  S:  a"  [        S[        R                  SS5      U l
        O![        S[        R                  SS5      U l
        U R
                  R                  S:X  a1  [        [        R                  * [        R                  SS5      U l        g [        S[        R                  SS5      U l        g rÎ   )r‰   r)   r   r¨   r   r   rÏ   r   r#   r$   r%   r&   rÐ   s      €r(   r)   Ú HalfTweedieLossIdentity.__init__d  sÝ   ø€ Ü‰ÑÜ+´%¸³,Ñ?Ü“ð 	ñ 	
ð �:‰:×Ñ˜qÓ Ü#+¬R¯V©V¨G´R·V±V¸UÀEÓ#JˆDÕ Ø�Z‰Z×Ñ Ó!Ü#+¨A¬r¯v©v°t¸UÓ#CˆDÕ ä#+¨A¬r¯v©v°u¸eÓ#DˆDÔ à�:‰:×Ñ˜qÓ Ü#+¬R¯V©V¨G´R·V±V¸UÀEÓ#JˆDÕ ä#+¨A¬r¯v©v°u¸eÓ#DˆDÕ r+   ©r&   r%   rÖ   r�   r�   s   @r(   rØ   rØ   H  s   ø† ñ÷6Eõ Er+   rØ   c                   ó@   ^ • \ rS rSrSrSU 4S jjrSS jrS rSrU =r	$ )ÚHalfBinomialLossiv  a	  Half Binomial deviance loss with logit link, for binary classification.

This is also know as binary cross entropy, log-loss and logistic loss.

Domain:
y_true in [0, 1], i.e. regression on the unit interval
y_pred in (0, 1), i.e. boundaries excluded

Link:
y_pred = expit(raw_prediction)

For a given sample x_i, half Binomial deviance is defined as the negative
log-likelihood of the Binomial/Bernoulli distribution and can be expressed
as::

    loss(x_i) = log(1 + exp(raw_pred_i)) - y_true_i * raw_pred_i

See The Elements of Statistical Learning, by Hastie, Tibshirani, Friedman,
section 4.4.1 (about logistic regression).

Note that the formulation works for classification, y = {0, 1}, as well as
logistic regression, y = [0, 1].
If you add `constant_to_optimal_zero` to the loss, you get half the
Bernoulli/binomial deviance.

More details: Inserting the predicted probability y_pred = expit(raw_prediction)
in the loss gives the well known::

    loss(x_i) = - y_true_i * log(y_pred_i) - (1 - y_true_i) * log(1 - y_pred_i)
c                 ój   >• [         TU ]  [        5       [        5       SS9  [	        SSSS5      U l        g ©Nr   ©r   r   r"   r   r   T)r‰   r)   r   r   r   r%   rŠ   s     €r(   r)   ÚHalfBinomialLoss.__init__–  s8   ø€ Ü‰ÑÜ$Ó&Ü“Øð 	ñ 	
ô
  (¨¨1¨d°DÓ9ˆÕr+   c                 óP   • [        X5      [        SU-
  SU-
  5      -   nUb  X2-  nU$ r{   r   rÀ   s       r(   rj   Ú)HalfBinomialLoss.constant_to_optimal_zerož  s3   € ä�VÓ$¤u¨Q°©Z¸¸V¹Ó'DÑDˆØÑ$ØÑ!ˆDØˆr+   c                 ó4  • UR                   S:X  a$  UR                  S   S:X  a  UR                  S5      n[        R                  " UR                  S   S4UR
                  S9nU R                  R                  U5      USS2S4'   SUSS2S4   -
  USS2S4'   U$ ©zõPredict probabilities.

Parameters
----------
raw_prediction : array of shape (n_samples,) or (n_samples, 1)
    Raw prediction values (in link space).

Returns
-------
proba : array of shape (n_samples, 2)
    Element-wise class probabilities.
r   r   r   rE   N©r?   r@   rA   r#   rs   rF   r   Úinverse©r'   r:   Úprobas      r(   Úpredict_probaÚHalfBinomialLoss.predict_proba¥  ó”   € ð ×Ñ !Ó#¨×(<Ñ(<¸QÑ(?À1Ó(DØ+×3Ñ3°AÓ6ˆNÜ—’˜.×.Ñ.¨qÑ1°1Ð5¸^×=QÑ=QÑRˆØ—i‘i×'Ñ'¨Ó7ˆŠa�ˆd‰Ø˜%¢ 1 ™+‘oˆŠa�ˆd‰Øˆr+   rÂ   ry   ©
r|   r}   r~   r   r€   r)   rj   rê   rƒ   rŽ   r�   s   @r(   rÝ   rÝ   v  s   ø† ñ÷>:ô÷ð r+   rÝ   c                   ó\   ^ • \ rS rSrSrSrS
U 4S jjrS rSS jrS r	    SS jr
S	rU =r$ )ÚHalfMultinomialLossi»  a4  Categorical cross-entropy loss, for multiclass classification.

Domain:
y_true in {0, 1, 2, 3, .., n_classes - 1}
y_pred has n_classes elements, each element in (0, 1)

Link:
y_pred = softmax(raw_prediction)

Note: We assume y_true to be already label encoded. The inverse link is
softmax. But the full link function is the symmetric multinomial logit
function.

For a given sample x_i, the categorical cross-entropy loss is defined as
the negative log-likelihood of the multinomial distribution, it
generalizes the binary cross-entropy to more than 2 classes::

    loss_i = log(sum(exp(raw_pred_{i, k}), k=0..n_classes-1))
            - sum(y_true_{i, k} * raw_pred_{i, k}, k=0..n_classes-1)

See [1].

Note that for the hessian, we calculate only the diagonal part in the
classes: If the full hessian for classes k and l and sample i is H_i_k_l,
we calculate H_i_k_k, i.e. k=l.

Reference
---------
.. [1] :arxiv:`Simon, Noah, J. Friedman and T. Hastie.
    "A Blockwise Descent Algorithm for Group-penalized Multiresponse and
    Multinomial Regression".
    <1311.6529>`
Tc                 ó¬   >• [         TU ]  [        5       [        5       US9  [	        S[
        R                  SS5      U l        [	        SSSS5      U l        g )Nrà   r   TFr   )	r‰   r)   r   r   r   r#   r$   r%   r&   )r'   r;   r"   r‹   s      €r(   r)   ÚHalfMultinomialLoss.__init__à  sP   ø€ Ü‰ÑÜ'Ó)Ü!Ó#Øð 	ñ 	
ô
  (¨¬2¯6©6°4¸Ó?ˆÔÜ'¨¨1¨e°UÓ;ˆÕr+   c                 óž   • U R                   R                  U5      =(       a,    [        R                  " UR	                  [
        5      U:H  5      $ r-   )r%   r.   r#   ÚallÚastypeÚintr/   s     r(   r1   Ú#HalfMultinomialLoss.in_y_true_rangeé  s6   € ð ×#Ñ#×,Ñ,¨QÓ/×N´B·F²F¸1¿8¹8ÄC»=ÈAÑ;MÓ4NÐNr+   c                 ó´  • [         R                  " U R                  UR                  S9n[         R                  " UR                  5      R
                  n[        U R                  5       H<  n[         R                  " X:H  USS9X5'   [         R                  " X5   USU-
  5      X5'   M>     U R                  R                  USSS24   5      R                  S5      $ )z�Compute raw_prediction of an intercept-only model.

This is the softmax of the weighted average of the target, i.e. over
the samples axis=0.
rE   r   rX   r   Néÿÿÿÿ)r#   Úzerosr"   rF   r[   r\   ÚrangerT   ra   r   Úreshape)r'   r9   r;   Úoutr\   Úks         r(   re   Ú&HalfMultinomialLoss.fit_intercept_onlyò  sŸ   € ô �hŠh�t—~‘~¨V¯\©\Ñ:ˆÜ�hŠh�v—|‘|Ó$×(Ñ(ˆÜ�t—~‘~Ö&ˆAÜ—Z’Z ¡°]ÈÑKˆC‰FÜ—W’W˜S™V S¨!¨c©'Ó2ˆC‹Fñ 'ð �y‰y�~‰~˜c $ª '™lÓ+×3Ñ3°BÓ7Ð7r+   c                 ó8   • U R                   R                  U5      $ )zõPredict probabilities.

Parameters
----------
raw_prediction : array of shape (n_samples, n_classes)
    Raw prediction values (in link space).

Returns
-------
proba : array of shape (n_samples, n_classes)
    Element-wise class probabilities.
)r   rç   )r'   r:   s     r(   rê   Ú!HalfMultinomialLoss.predict_probaÿ  s   € ð �y‰y× Ñ  Ó0Ð0r+   c           	      ó
  • UcG  Uc-  [         R                  " U5      n[         R                  " U5      nO0[         R                  " U5      nOUc  [         R                  " U5      nU R                  R                  UUUUUUS9  XE4$ )a“  Compute gradient and class probabilities fow raw_prediction.

Parameters
----------
y_true : C-contiguous array of shape (n_samples,)
    Observed, true target values.
raw_prediction : array of shape (n_samples, n_classes)
    Raw prediction values (in link space).
sample_weight : None or C-contiguous array of shape (n_samples,)
    Sample weights.
gradient_out : None or array of shape (n_samples, n_classes)
    A location into which the gradient is stored. If None, a new array
    might be created.
proba_out : None or array of shape (n_samples, n_classes)
    A location into which the class probabilities are stored. If None,
    a new array might be created.
n_threads : int, default=1
    Might use openmp thread parallelism.

Returns
-------
gradient : array of shape (n_samples, n_classes)
    Element-wise gradients.

proba : array of shape (n_samples, n_classes)
    Element-wise class probabilities.
)r9   r:   r;   rG   Ú	proba_outr=   )r#   r>   r   Úgradient_proba)r'   r9   r:   r;   rG   r  r=   s          r(   r  Ú"HalfMultinomialLoss.gradient_proba  sƒ   € ðH ÑØÑ Ü!Ÿ}š}¨^Ó<�ÜŸMšM¨.Ó9‘	ä!Ÿ}š}¨YÓ7‘ØÑÜŸš lÓ3ˆIà�
‰
×!Ñ!ØØ)Ø'Ø%ØØð 	"ñ 	
ð Ð&Ð&r+   rÛ   )Né   ry   rz   )r|   r}   r~   r   r€   rr   r)   r1   re   rê   r  rƒ   rŽ   r�   s   @r(   rï   rï   »  s=   ø† ñ ðD €M÷<òOô8ò1ð& ØØØ÷5'ò 5'r+   rï   c                   ó@   ^ • \ rS rSrSrSU 4S jjrSS jrS rSrU =r	$ )ÚExponentialLossiF  aÂ  Exponential loss with (half) logit link, for binary classification.

This is also know as boosting loss.

Domain:
y_true in [0, 1], i.e. regression on the unit interval
y_pred in (0, 1), i.e. boundaries excluded

Link:
y_pred = expit(2 * raw_prediction)

For a given sample x_i, the exponential loss is defined as::

    loss(x_i) = y_true_i * exp(-raw_pred_i)) + (1 - y_true_i) * exp(raw_pred_i)

See:
- J. Friedman, T. Hastie, R. Tibshirani.
  "Additive logistic regression: a statistical view of boosting (With discussion
  and a rejoinder by the authors)." Ann. Statist. 28 (2) 337 - 407, April 2000.
  https://doi.org/10.1214/aos/1016218223
- A. Buja, W. Stuetzle, Y. Shen. (2005).
  "Loss Functions for Binary Class Probability Estimation and Classification:
  Structure and Applications."

Note that the formulation works for classification, y = {0, 1}, as well as
"exponential logistic" regression, y = [0, 1].
Note that this is a proper scoring rule, but without it's canonical link.

More details: Inserting the predicted probability
y_pred = expit(2 * raw_prediction) in the loss gives::

    loss(x_i) = y_true_i * sqrt((1 - y_pred_i) / y_pred_i)
        + (1 - y_true_i) * sqrt(y_pred_i / (1 - y_pred_i))
c                 ój   >• [         TU ]  [        5       [        5       SS9  [	        SSSS5      U l        g rß   )r‰   r)   r
   r   r   r%   rŠ   s     €r(   r)   ÚExponentialLoss.__init__j  s8   ø€ Ü‰ÑÜ#Ó%Ü“Øð 	ñ 	
ô
  (¨¨1¨d°DÓ9ˆÕr+   c                 óR   • S[         R                  " USU-
  -  5      -  nUb  X2-  nU$ )Néþÿÿÿr   )r#   ÚsqrtrÀ   s       r(   rj   Ú(ExponentialLoss.constant_to_optimal_zeror  s1   € à”B—G’G˜F a¨&¡jÑ1Ó2Ñ2ˆØÑ$ØÑ!ˆDØˆr+   c                 ó4  • UR                   S:X  a$  UR                  S   S:X  a  UR                  S5      n[        R                  " UR                  S   S4UR
                  S9nU R                  R                  U5      USS2S4'   SUSS2S4   -
  USS2S4'   U$ rå   ræ   rè   s      r(   rê   ÚExponentialLoss.predict_probay  rì   r+   rÂ   ry   rí   r�   s   @r(   r  r  F  s   ø† ñ!÷F:ô÷ð r+   r  )
Úsquared_errorÚabsolute_errorÚpinball_lossÚ
huber_lossÚpoisson_lossÚ
gamma_lossÚtweedie_lossÚbinomial_lossÚmultinomial_lossÚexponential_loss)*r€   r¦   Únumpyr#   Úscipy.specialr   Úutilsr   Úutils.statsr   Ú_lossr	   r
   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r   r†   r‘   r�   r°   r¼   rÅ   rÌ   rØ   rÝ   rï   r  Ú_LOSSESr„   r+   r(   Ú<module>r      s  ðñó$ ã Ý å  Ý .÷÷ ÷ ñ ÷÷ ÷8z!ñ z!ôB6�xô 6ô.#C�Hô #CôL<�(ô <ô~F@�ô F@ôR�hô ô@�Hô ô>=�hô =ô@+E˜hô +Eô\B�xô BôJH'˜(ô H'ôVF�hô FðT &Ø#ØØØ#ØØ#Ø%Ø+Ø'ñ�r+   