U
    <¼|e¿e  ã                   @   s6   d Z ddlZddlmZ ddlmZ G dd„ dƒZdS )zA
Loss functions for linear models with raw_prediction = X @ coef
é    N)Úsparseé   )Úsquared_normc                   @   sl   e Zd ZdZdd„ Zddd„Zdd„ Zd	d
„ Zdd„ Zddd„Z	ddd„Z
ddd„Zddd„Zddd„ZdS )ÚLinearModelLossa…  General class for loss functions with raw_prediction = X @ coef + intercept.

    Note that raw_prediction is also known as linear predictor.

    The loss is the sum of per sample losses and includes a term for L2
    regularization::

        loss = sum_i s_i loss(y_i, X_i @ coef + intercept)
               + 1/2 * l2_reg_strength * ||coef||_2^2

    with sample weights s_i=1 if sample_weight=None.

    Gradient and hessian, for simplicity without intercept, are::

        gradient = X.T @ loss.gradient + l2_reg_strength * coef
        hessian = X.T @ diag(loss.hessian) @ X + l2_reg_strength * identity

    Conventions:
        if fit_intercept:
            n_dof =  n_features + 1
        else:
            n_dof = n_features

        if base_loss.is_multiclass:
            coef.shape = (n_classes, n_dof) or ravelled (n_classes * n_dof,)
        else:
            coef.shape = (n_dof,)

        The intercept term is at the end of the coef array:
        if base_loss.is_multiclass:
            if coef.shape (n_classes, n_dof):
                intercept = coef[:, -1]
            if coef.shape (n_classes * n_dof,)
                intercept = coef[n_features::n_dof] = coef[(n_dof-1)::n_dof]
            intercept.shape = (n_classes,)
        else:
            intercept = coef[-1]

    Note: If coef has shape (n_classes * n_dof,), the 2d-array can be reconstructed as

        coef.reshape((n_classes, -1), order="F")

    The option order="F" makes coef[:, i] contiguous. This, in turn, makes the
    coefficients without intercept, coef[:, :-1], contiguous and speeds up
    matrix-vector computations.

    Note: If the average loss per sample is wanted instead of the sum of the loss per
    sample, one can simply use a rescaled sample_weight such that
    sum(sample_weight) = 1.

    Parameters
    ----------
    base_loss : instance of class BaseLoss from sklearn._loss.
    fit_intercept : bool
    c                 C   s   || _ || _d S )N)Ú	base_lossÚfit_intercept)Úselfr   r   © r	   ú^/var/www/website-v5/atlas_env/lib/python3.8/site-packages/sklearn/linear_model/_linear_loss.pyÚ__init__B   s    zLinearModelLoss.__init__Nc                 C   sZ   |j d }| jj}| jr"|d }n|}| jjrFtj|||f|dd�}ntj|||d�}|S )aâ  Allocate coef of correct shape with zeros.

        Parameters:
        -----------
        X : {array-like, sparse matrix} of shape (n_samples, n_features)
            Training data.
        dtype : data-type, default=None
            Overrides the data type of coef. With dtype=None, coef will have the same
            dtype as X.

        Returns
        -------
        coef : ndarray of shape (n_dof,) or (n_classes, n_dof)
            Coefficients of a linear model.
        é   ÚF)ÚshapeÚdtypeÚorder©r   r   )r   r   Ú	n_classesr   Úis_multiclassÚnpÚ
zeros_like)r   ÚXr   Ú
n_featuresr   Ún_dofÚcoefr	   r	   r
   Úinit_zero_coefF   s    

zLinearModelLoss.init_zero_coefc                 C   sŒ   | j js.| jr$|d }|dd… }q„d}|}nV|jdkrP|j| j jdfdd�}n|}| jr€|dd…df }|dd…dd…f }nd}||fS )a˜  Helper function to get coefficients and intercept.

        Parameters
        ----------
        coef : ndarray of shape (n_dof,), (n_classes, n_dof) or (n_classes * n_dof,)
            Coefficients of a linear model.
            If shape (n_classes * n_dof,), the classes of one feature are contiguous,
            i.e. one reconstructs the 2d-array via
            coef.reshape((n_classes, -1), order="F").

        Returns
        -------
        weights : ndarray of shape (n_features,) or (n_classes, n_features)
            Coefficients without intercept term.
        intercept : float or ndarray of shape (n_classes,)
            Intercept terms.
        éÿÿÿÿNç        r   r   ©r   )r   r   r   ÚndimÚreshaper   )r   r   Ú	interceptÚweightsr	   r	   r
   Úweight_interceptb   s    
z LinearModelLoss.weight_interceptc                 C   s<   |   |¡\}}| jjs$|| | }n||j | }|||fS )ai  Helper function to get coefficients, intercept and raw_prediction.

        Parameters
        ----------
        coef : ndarray of shape (n_dof,), (n_classes, n_dof) or (n_classes * n_dof,)
            Coefficients of a linear model.
            If shape (n_classes * n_dof,), the classes of one feature are contiguous,
            i.e. one reconstructs the 2d-array via
            coef.reshape((n_classes, -1), order="F").
        X : {array-like, sparse matrix} of shape (n_samples, n_features)
            Training data.

        Returns
        -------
        weights : ndarray of shape (n_features,) or (n_classes, n_features)
            Coefficients without intercept term.
        intercept : float or ndarray of shape (n_classes,)
            Intercept terms.
        raw_prediction : ndarray of shape (n_samples,) or             (n_samples, n_classes)
        )r"   r   r   ÚT)r   r   r   r!   r    Úraw_predictionr	   r	   r
   Úweight_intercept_raw‰   s
    z$LinearModelLoss.weight_intercept_rawc                 C   s&   |j dkr|| nt|ƒ}d| | S )z5Compute L2 penalty term l2_reg_strength/2 *||w||_2^2.r   g      à?)r   r   )r   r!   Úl2_reg_strengthZnorm2_wr	   r	   r
   Ú
l2_penalty©   s    zLinearModelLoss.l2_penaltyr   r   c                 C   sV   |dkr|   ||¡\}}	}n|  |¡\}}	| jj||||d�}
|
 ¡ }
|
|  ||¡ S )aô  Compute the loss as sum over point-wise losses.

        Parameters
        ----------
        coef : ndarray of shape (n_dof,), (n_classes, n_dof) or (n_classes * n_dof,)
            Coefficients of a linear model.
            If shape (n_classes * n_dof,), the classes of one feature are contiguous,
            i.e. one reconstructs the 2d-array via
            coef.reshape((n_classes, -1), order="F").
        X : {array-like, sparse matrix} of shape (n_samples, n_features)
            Training data.
        y : contiguous array of shape (n_samples,)
            Observed, true target values.
        sample_weight : None or contiguous array of shape (n_samples,), default=None
            Sample weights.
        l2_reg_strength : float, default=0.0
            L2 regularization strength
        n_threads : int, default=1
            Number of OpenMP threads to use.
        raw_prediction : C-contiguous array of shape (n_samples,) or array of             shape (n_samples, n_classes)
            Raw prediction values (in link space). If provided, these are used. If
            None, then raw_prediction = X @ coef + intercept is calculated.

        Returns
        -------
        loss : float
            Sum of losses per sample plus penalty.
        N©Úy_truer$   Úsample_weightÚ	n_threads)r%   r"   r   ÚlossÚsumr'   )r   r   r   Úyr*   r&   r+   r$   r!   r    r,   r	   r	   r
   r,   ®   s    'üzLinearModelLoss.lossc                 C   s:  |j d | jj }}	|t| jƒ }
|dkr>|  ||¡\}}}n|  |¡\}}| jj||||d�\}}| ¡ }||  	||¡7 }| jj
sÂtj||jd�}|j| ||  |d|…< | jrÀ| ¡ |d< nptj|	|
f|jdd�}|j| ||  |dd…d|…f< | j�r|jdd	�|dd…df< |jdk�r2|jdd
�}||fS )aN  Computes the sum of loss and gradient w.r.t. coef.

        Parameters
        ----------
        coef : ndarray of shape (n_dof,), (n_classes, n_dof) or (n_classes * n_dof,)
            Coefficients of a linear model.
            If shape (n_classes * n_dof,), the classes of one feature are contiguous,
            i.e. one reconstructs the 2d-array via
            coef.reshape((n_classes, -1), order="F").
        X : {array-like, sparse matrix} of shape (n_samples, n_features)
            Training data.
        y : contiguous array of shape (n_samples,)
            Observed, true target values.
        sample_weight : None or contiguous array of shape (n_samples,), default=None
            Sample weights.
        l2_reg_strength : float, default=0.0
            L2 regularization strength
        n_threads : int, default=1
            Number of OpenMP threads to use.
        raw_prediction : C-contiguous array of shape (n_samples,) or array of             shape (n_samples, n_classes)
            Raw prediction values (in link space). If provided, these are used. If
            None, then raw_prediction = X @ coef + intercept is calculated.

        Returns
        -------
        loss : float
            Sum of losses per sample plus penalty.

        gradient : ndarray of shape coef.shape
             The gradient of the loss.
        r   Nr(   ©r   r   r   ©r   r   r   ©Úaxisr   )r   r   r   Úintr   r%   r"   Úloss_gradientr-   r'   r   r   Ú
empty_liker   r#   Úemptyr   Úravel)r   r   r   r.   r*   r&   r+   r$   r   r   r   r!   r    r,   Úgrad_pointwiseÚgradr	   r	   r
   r4   ä   s2    *ü
"zLinearModelLoss.loss_gradientc                 C   s   |j d | jj }}	|t| jƒ }
|dkr>|  ||¡\}}}n|  |¡\}}| jj||||d�}| jjs¨t	j
||jd�}|j| ||  |d|…< | jr¤| ¡ |d< |S t	j|	|
f|jdd�}|j| ||  |dd…d|…f< | j�r |jdd	�|dd…df< |jdk�r|jdd
�S |S dS )aõ  Computes the gradient w.r.t. coef.

        Parameters
        ----------
        coef : ndarray of shape (n_dof,), (n_classes, n_dof) or (n_classes * n_dof,)
            Coefficients of a linear model.
            If shape (n_classes * n_dof,), the classes of one feature are contiguous,
            i.e. one reconstructs the 2d-array via
            coef.reshape((n_classes, -1), order="F").
        X : {array-like, sparse matrix} of shape (n_samples, n_features)
            Training data.
        y : contiguous array of shape (n_samples,)
            Observed, true target values.
        sample_weight : None or contiguous array of shape (n_samples,), default=None
            Sample weights.
        l2_reg_strength : float, default=0.0
            L2 regularization strength
        n_threads : int, default=1
            Number of OpenMP threads to use.
        raw_prediction : C-contiguous array of shape (n_samples,) or array of             shape (n_samples, n_classes)
            Raw prediction values (in link space). If provided, these are used. If
            None, then raw_prediction = X @ coef + intercept is calculated.

        Returns
        -------
        gradient : ndarray of shape coef.shape
             The gradient of the loss.
        r   Nr(   r/   r   r   r0   r   r1   r   )r   r   r   r3   r   r%   r"   Úgradientr   r   r5   r   r#   r-   r6   r   r7   )r   r   r   r.   r*   r&   r+   r$   r   r   r   r!   r    r8   r9   r	   r	   r
   r:   /  s0    'ü"zLinearModelLoss.gradientc
                 C   sê  |j \}
}|t| jƒ }|	dkr4|  ||¡\}}}	n|  |¡\}}| jj||	||d�\}}t |dk¡dk}t 	|¡}| jj
�sÜ|dkrštj||jd�}n|}|j| ||  |d|…< | jrÊ| ¡ |d< |dkrètj||f|jd�}n|}|rú|||fS t |¡�r<|jtj|df|
|
fd� |  ¡ |d|…d|…f< n2|dd…df | }t |j|¡|d|…d|…f< |dk�rœ| d¡d|| |d	 …  |7  < | j�rà|j| }||dd…df< ||ddd…f< | ¡ |d
< nt‚|||fS )aå  Computes gradient and hessian w.r.t. coef.

        Parameters
        ----------
        coef : ndarray of shape (n_dof,), (n_classes, n_dof) or (n_classes * n_dof,)
            Coefficients of a linear model.
            If shape (n_classes * n_dof,), the classes of one feature are contiguous,
            i.e. one reconstructs the 2d-array via
            coef.reshape((n_classes, -1), order="F").
        X : {array-like, sparse matrix} of shape (n_samples, n_features)
            Training data.
        y : contiguous array of shape (n_samples,)
            Observed, true target values.
        sample_weight : None or contiguous array of shape (n_samples,), default=None
            Sample weights.
        l2_reg_strength : float, default=0.0
            L2 regularization strength
        n_threads : int, default=1
            Number of OpenMP threads to use.
        gradient_out : None or ndarray of shape coef.shape
            A location into which the gradient is stored. If None, a new array
            might be created.
        hessian_out : None or ndarray
            A location into which the hessian is stored. If None, a new array
            might be created.
        raw_prediction : C-contiguous array of shape (n_samples,) or array of             shape (n_samples, n_classes)
            Raw prediction values (in link space). If provided, these are used. If
            None, then raw_prediction = X @ coef + intercept is calculated.

        Returns
        -------
        gradient : ndarray of shape coef.shape
             The gradient of the loss.

        hessian : ndarray
            Hessian matrix.

        hessian_warning : bool
            True if pointwise hessian has more than half of its elements non-positive.
        Nr(   r   g      Ð?r/   r   r   ©r   r   )r   r   )r   r3   r   r%   r"   r   Úgradient_hessianr   ÚmeanÚabsr   r5   r   r#   r-   r6   r   ÚissparseÚ
dia_matrixÚtoarrayÚdotr   ÚNotImplementedError)r   r   r   r.   r*   r&   r+   Úgradient_outÚhessian_outr$   Ú	n_samplesr   r   r!   r    r8   Úhess_pointwiseÚhessian_warningr9   ÚhessZWXZXhr	   r	   r
   r<   v  sf    5
ü




 ÿÿüÿ


 ÿþ
z LinearModelLoss.gradient_hessianc              
      sÌ  ˆ j ˆjj \}‰‰ˆtˆjƒ ‰ˆ ˆˆ ¡\‰}}	ˆjj�sˆjj||	ˆ
|d�\}
}tj	ˆˆj
d�}ˆ j|
 ˆˆ  |dˆ…< ˆjr’|
 ¡ |d< | ¡ ‰t ˆ ¡rÀtj|df||fd�ˆ  ‰n|dd…tjf ˆ  ‰ˆj�r t t ˆjdd�¡¡‰t ˆ¡‰‡ ‡‡‡‡‡‡fdd	„}nªˆjj||	ˆ
|d�\}
‰	tjˆˆfˆj
d
d�}|
jˆ  ˆˆ  |dd…dˆ…f< ˆj�rŠ|
jdd�|dd…df< ‡ ‡‡‡‡‡‡	‡
‡‡f
dd	„}ˆjdk�rÄ|jd
d�|fS ||fS )a¡  Computes gradient and hessp (hessian product function) w.r.t. coef.

        Parameters
        ----------
        coef : ndarray of shape (n_dof,), (n_classes, n_dof) or (n_classes * n_dof,)
            Coefficients of a linear model.
            If shape (n_classes * n_dof,), the classes of one feature are contiguous,
            i.e. one reconstructs the 2d-array via
            coef.reshape((n_classes, -1), order="F").
        X : {array-like, sparse matrix} of shape (n_samples, n_features)
            Training data.
        y : contiguous array of shape (n_samples,)
            Observed, true target values.
        sample_weight : None or contiguous array of shape (n_samples,), default=None
            Sample weights.
        l2_reg_strength : float, default=0.0
            L2 regularization strength
        n_threads : int, default=1
            Number of OpenMP threads to use.

        Returns
        -------
        gradient : ndarray of shape coef.shape
             The gradient of the loss.

        hessp : callable
            Function that takes in a vector input of shape of gradient and
            and returns matrix-vector product with hessian.
        r(   r/   Nr   r   r;   r1   c                    s¾   t  | ¡}t ˆ ¡r4ˆ jˆ| d ˆ…   |d ˆ…< n$t j ˆ jˆ| d ˆ… g¡|d ˆ…< |d ˆ…  ˆ| d ˆ…  7  < ˆjrº|d ˆ…  | d ˆ 7  < ˆ| d ˆ…  ˆ| d   |d< |S )Nr   )r   r5   r   r?   r#   ÚlinalgÚ	multi_dotr   )ÚsÚret)r   ÚhXÚhX_sumÚhessian_sumr&   r   r   r	   r
   ÚhesspD  s    

 $  z7LinearModelLoss.gradient_hessian_product.<locals>.hesspr   r0   c                    s  | j ˆdfdd�} ˆjr>| d d …df }| d d …d d…f } nd}ˆ | j | }|ˆ | jdd�d d …tjf 7 }|ˆ9 }ˆd k	rš|ˆd d …tjf 9 }tjˆˆfˆ	jdd�}|jˆ  ˆ|   |d d …d ˆ…f< ˆjrð|jdd�|d d …df< ˆjdk�r|j	dd�S |S d S )Nr   r   r   r   r   r1   r0   )
r   r   r#   r-   r   Únewaxisr6   r   r   r7   )rL   Zs_interceptÚtmpZ	hess_prod)
r   r   r&   r   r   r   Úprobar*   r   r!   r	   r
   rQ   w  s"    $"r   r   )r   r   r   r3   r   r%   r   r<   r   r5   r   r#   r-   r   r?   r@   rR   ÚsqueezeÚasarrayÚ
atleast_1dÚgradient_probar6   r   r7   )r   r   r   r.   r*   r&   r+   rF   r    r$   r8   rG   r9   rQ   r	   )r   r   rN   rO   rP   r&   r   r   r   rT   r*   r   r!   r
   Úgradient_hessian_productþ  sN     
ü

ÿÿ
ü
"z(LinearModelLoss.gradient_hessian_product)N)Nr   r   N)Nr   r   N)Nr   r   N)Nr   r   NNN)Nr   r   )Ú__name__Ú
__module__Ú__qualname__Ú__doc__r   r   r"   r%   r'   r,   r4   r:   r<   rY   r	   r	   r	   r
   r   	   sB   8
' 
    ø
;    ø
P    ø
L      ö
 
     ÿr   )r]   Únumpyr   Úscipyr   Úutils.extmathr   r   r	   r	   r	   r
   Ú<module>   s   