U
    Ãmœd¢V  ã                   @   s°   d dl mZ d dlmZmZmZ d dlmZ d dlm	Z	 d dl
Zddd„Zddd	„Zdd
d„Zdd„ Zdd„ Zddd„Zddd„Zi fdd„ZG dd„ dƒZG dd„ deƒZdS )é    )ÚRegularizedResults)Ú_calc_nodewise_rowÚ_calc_nodewise_weightÚ_calc_approx_inv_cov)ÚLikelihoodModelResults)ÚOLSNc                 C   s   |dkrt dƒ‚| jf |ŽjS )aÂ  estimates the regularized fitted parameters.

    Parameters
    ----------
    mod : statsmodels model class instance
        The model for the current partition.
    pnum : scalar
        Index of current partition
    partitions : scalar
        Total number of partitions
    fit_kwds : dict-like or None
        Keyword arguments to be given to fit_regularized

    Returns
    -------
    An array of the parameters for the regularized fit
    NzD_est_regularized_naive currently requires that fit_kwds not be None.)Ú
ValueErrorÚfit_regularizedÚparams©ÚmodÚpnumÚ
partitionsÚfit_kwds© r   ú`/home/sam/Atlas/atlas_env/lib/python3.8/site-packages/statsmodels/base/distributed_estimation.pyÚ_est_regularized_naiveK   s    r   c                 C   s   |dkrt dƒ‚| jf |ŽjS )a¬  estimates the unregularized fitted parameters.

    Parameters
    ----------
    mod : statsmodels model class instance
        The model for the current partition.
    pnum : scalar
        Index of current partition
    partitions : scalar
        Total number of partitions
    fit_kwds : dict-like or None
        Keyword arguments to be given to fit

    Returns
    -------
    An array of the parameters for the fit
    NzF_est_unregularized_naive currently requires that fit_kwds not be None.)r   Úfitr
   r   r   r   r   Ú_est_unregularized_naivee   s    r   c                 C   sN   t | d ƒ}t | ƒ}t |¡}| D ]}||7 }q"|| }d|t |¡|k < |S )a   joins the results from each run of _est_<type>_naive
    and returns the mean estimate of the coefficients

    Parameters
    ----------
    params_l : list
        A list of arrays of coefficients.
    threshold : scalar
        The threshold at which the coefficients will be cut.
    r   )ÚlenÚnpÚzerosÚabs)Zparams_lÚ	thresholdÚpr   Ú	params_mnr
   r   r   r   Ú_join_naive   s    

r   c                 C   s*   | j t |¡f|Ž }||d|  7 }|S )a  calculates the log-likelihood gradient for the debiasing

    Parameters
    ----------
    mod : statsmodels model class instance
        The model for the current partition.
    params : array_like
        The estimated coefficients for the current partition.
    alpha : scalar or array_like
        The penalty weight.  If a scalar, the same penalty weight
        applies to all variables in the model.  If a vector, it
        must have the same length as `params`, and contains a
        penalty weight for each coefficient.
    L1_wt : scalar
        The fraction of the penalty given to the L1 penalty term.
        Must be between 0 and 1 (inclusive).  If 0, the fit is
        a ridge fit, if 1 it is a lasso fit.
    score_kwds : dict-like or None
        Keyword arguments for the score function.

    Returns
    -------
    An array-like object of the same dimension as params

    Notes
    -----
    In general:

    gradient l_k(params)

    where k corresponds to the index of the partition

    For OLS:

    X^T(y - X^T params)
    é   )Zscorer   Úasarray)r   r
   ÚalphaÚL1_wtÚ
score_kwdsÚgradr   r   r   Ú
_calc_grad˜   s    &r#   c                 C   s0   t  | jt  |¡f|Ž¡}|dd…df | j S )aú  calculates the weighted design matrix necessary to generate
    the approximate inverse covariance matrix

    Parameters
    ----------
    mod : statsmodels model class instance
        The model for the current partition.
    params : array_like
        The estimated coefficients for the current partition.
    hess_kwds : dict-like or None
        Keyword arguments for the hessian function.

    Returns
    -------
    An array-like object, updated design matrix, same dimension
    as mod.exog
    N)r   ÚsqrtZhessian_factorr   Úexog)r   r
   Ú	hess_kwdsZrhessr   r   r   Ú_calc_wdesign_matÃ   s    r'   c                 C   s  |dkri n|}|dkri n|}|dkr2t dƒ‚n|d }d|krL|d }nd}| jj\}}	tt d|	 | ¡ƒ}
| jf |Žj}t| ||||ƒ| }t	| ||ƒ}g }g }t
||
 t|d |
 |	ƒƒD ]2}t|||ƒ}| |¡ t||||ƒ}| |¡ qÄ||||fS )a‚  estimates the regularized fitted parameters, is the default
    estimation_method for class DistributedModel.

    Parameters
    ----------
    mod : statsmodels model class instance
        The model for the current partition.
    mnum : scalar
        Index of current partition.
    partitions : scalar
        Total number of partitions.
    fit_kwds : dict-like or None
        Keyword arguments to be given to fit_regularized
    score_kwds : dict-like or None
        Keyword arguments for the score function.
    hess_kwds : dict-like or None
        Keyword arguments for the Hessian function.

    Returns
    -------
    A tuple of parameters for regularized fit
        An array-like object of the fitted parameters, params
        An array-like object for the gradient
        A list of array like objects for nodewise_row
        A list of array like objects for nodewise_weight
    NzG_est_regularized_debiased currently requires that fit_kwds not be None.r   r    r   g      ð?)r   r%   ÚshapeÚintr   Úceilr	   r
   r#   r'   ÚrangeÚminr   Úappendr   )r   Zmnumr   r   r!   r&   r   r    Znobsr   Zp_partr
   r"   ZwexogÚnodewise_row_lÚnodewise_weight_lÚidxZnodewise_rowZnodewise_weightr   r   r   Ú_est_regularized_debiasedÚ   s.    

 
ÿr1   c                 C   sÈ   t | d d ƒ}t | ƒ}t |¡}t |¡}g }g }| D ]8}||d 7 }||d 7 }| |d ¡ | |d ¡ q8t |¡}t |¡}|| }|d| 9 }t||ƒ}	||	 |¡ }
d|
t |
¡|k < |
S )a†  joins the results from each run of _est_regularized_debiased
    and returns the debiased estimate of the coefficients

    Parameters
    ----------
    results_l : list
        A list of tuples each one containing the params, grad,
        nodewise_row and nodewise_weight values for each partition.
    threshold : scalar
        The threshold at which the coefficients will be cut.
    r   r   é   é   g      ð¿)r   r   r   ÚextendÚarrayr   Údotr   )Ú	results_lr   r   r   r   Zgrad_mnr.   r/   ÚrZapprox_inv_covZdebiased_paramsr   r   r   Ú_join_debiased  s&    




r9   c           	      C   sF   | j  ¡ }| |¡ | j||f|Ž}| j||| jfd|i| j—Ž}|S )aà  handles the model fitting for each machine. NOTE: this
    is primarily handled outside of DistributedModel because
    joblib cannot handle class methods.

    Parameters
    ----------
    self : DistributedModel class instance
        An instance of DistributedModel.
    pnum : scalar
        index of current partition.
    endog : array_like
        endogenous data for current partition.
    exog : array_like
        exogenous data for current partition.
    fit_kwds : dict-like
        Keywords needed for the model fitting.
    init_kwds_e : dict-like
        Additional init_kwds to add for each partition.

    Returns
    -------
    estimation_method result.  For the default,
    _est_regularized_debiased, a tuple.
    r   )Ú	init_kwdsÚcopyÚupdateÚmodel_classÚestimation_methodr   Úestimation_kwds)	Úselfr   Úendogr%   r   Úinit_kwds_eZtemp_init_kwdsÚmodelÚresultsr   r   r   Ú_helper_fit_partitionH  s    

ÿþrE   c                   @   s8   e Zd ZdZddd„Zddd„Zddd	„Zdd
d„ZdS )ÚDistributedModelaÇ  
    Distributed model class

    Parameters
    ----------
    partitions : scalar
        The number of partitions that the data will be split into.
    model_class : statsmodels model class
        The model class which will be used for estimation. If None
        this defaults to OLS.
    init_kwds : dict-like or None
        Keywords needed for initializing the model, in addition to
        endog and exog.
    init_kwds_generator : generator or None
        Additional keyword generator that produces model init_kwds
        that may vary based on data partition.  The current usecase
        is for WLS and GLS
    estimation_method : function or None
        The method that performs the estimation for each partition.
        If None this defaults to _est_regularized_debiased.
    estimation_kwds : dict-like or None
        Keywords to be passed to estimation_method.
    join_method : function or None
        The method used to recombine the results from each partition.
        If None this defaults to _join_debiased.
    join_kwds : dict-like or None
        Keywords to be passed to join_method.
    results_class : results class or None
        The class of results that should be returned.  If None this
        defaults to RegularizedResults.
    results_kwds : dict-like or None
        Keywords to be passed to results class.

    Attributes
    ----------
    partitions : scalar
        See Parameters.
    model_class : statsmodels model class
        See Parameters.
    init_kwds : dict-like
        See Parameters.
    init_kwds_generator : generator or None
        See Parameters.
    estimation_method : function
        See Parameters.
    estimation_kwds : dict-like
        See Parameters.
    join_method : function
        See Parameters.
    join_kwds : dict-like
        See Parameters.
    results_class : results class
        See Parameters.
    results_kwds : dict-like
        See Parameters.

    Notes
    -----

    Examples
    --------
    Nc
           
      C   sº   || _ |d krt| _n|| _|d kr,i | _n|| _|d krBt| _n|| _|d krXi | _n|| _|d krnt| _n|| _|d kr„i | _	n|| _	|d kršt
| _n|| _|	d kr°i | _n|	| _d S ©N)r   r   r=   r:   r1   r>   r?   r9   Újoin_methodÚ	join_kwdsr   Úresults_classÚresults_kwds)
r@   r   r=   r:   r>   r?   rH   rI   rJ   rK   r   r   r   Ú__init__­  s2    zDistributedModel.__init__Ú
sequentialc           	      C   s‚   |dkri }|dkr$|   |||¡}n&|dkr>|  ||||¡}ntd| ƒ‚| j|f| jŽ}| jdgdgf| jŽ}| j||f| jŽS )ae  Performs the distributed estimation using the corresponding
        DistributedModel

        Parameters
        ----------
        data_generator : generator
            A generator that produces a sequence of tuples where the first
            element in the tuple corresponds to an endog array and the
            element corresponds to an exog array.
        fit_kwds : dict-like or None
            Keywords needed for the model fitting.
        parallel_method : str
            type of distributed estimation to be used, currently
            "sequential", "joblib" and "dask" are supported.
        parallel_backend : None or joblib parallel_backend object
            used to allow support for more complicated backends,
            ex: dask.distributed
        init_kwds_generator : generator or None
            Additional keyword generator that produces model init_kwds
            that may vary based on data partition.  The current usecase
            is for WLS and GLS

        Returns
        -------
        join_method result.  For the default, _join_debiased, it returns a
        p length array.
        NrM   Zjoblibz.parallel_method: %s is currently not supportedr   )	Úfit_sequentialÚ
fit_joblibr   rH   rI   r=   r:   rJ   rK   )	r@   Údata_generatorr   Zparallel_methodÚparallel_backendÚinit_kwds_generatorr7   r
   Zres_modr   r   r   r   Ü  s"    ÿþÿzDistributedModel.fitc                 C   s‚   g }|dkr>t |ƒD ]&\}\}}t| ||||ƒ}| |¡ qn@t t||ƒƒ}	|	D ],\}\\}}}
t| |||||
ƒ}| |¡ qP|S )a*  Sequentially performs the distributed estimation using
        the corresponding DistributedModel

        Parameters
        ----------
        data_generator : generator
            A generator that produces a sequence of tuples where the first
            element in the tuple corresponds to an endog array and the
            element corresponds to an exog array.
        fit_kwds : dict-like
            Keywords needed for the model fitting.
        init_kwds_generator : generator or None
            Additional keyword generator that produces model init_kwds
            that may vary based on data partition.  The current usecase
            is for WLS and GLS

        Returns
        -------
        join_method result.  For the default, _join_debiased, it returns a
        p length array.
        N)Ú	enumeraterE   r-   Úzip)r@   rP   r   rR   r7   r   rA   r%   rD   Útup_genrB   r   r   r   rN     s"    
ÿÿ
 ÿzDistributedModel.fit_sequentialc           
   	      s  ddl m} |tˆjƒ\}‰ }|dkrN|dkrN|‡ ‡‡fdd„t|ƒD ƒƒ}nÆ|dk	rŽ|dkrŽ|�$ |‡ ‡‡fdd„t|ƒD ƒƒ}W 5 Q R X n†|dkrÈ|dk	rÈtt||ƒƒ}	|‡ ‡‡fdd„|	D ƒƒ}nL|dk	�r|dk	�rtt||ƒƒ}	|�  |‡ ‡‡fdd„|	D ƒƒ}W 5 Q R X |S )	a©  Performs the distributed estimation in parallel using joblib

        Parameters
        ----------
        data_generator : generator
            A generator that produces a sequence of tuples where the first
            element in the tuple corresponds to an endog array and the
            element corresponds to an exog array.
        fit_kwds : dict-like
            Keywords needed for the model fitting.
        parallel_backend : None or joblib parallel_backend object
            used to allow support for more complicated backends,
            ex: dask.distributed
        init_kwds_generator : generator or None
            Additional keyword generator that produces model init_kwds
            that may vary based on data partition.  The current usecase
            is for WLS and GLS

        Returns
        -------
        join_method result.  For the default, _join_debiased, it returns a
        p length array.
        r   )Úparallel_funcNc                 3   s&   | ]\}\}}ˆ ˆ|||ˆƒV  qd S rG   r   ©Ú.0r   rA   r%   ©Úfr   r@   r   r   Ú	<genexpr>c  s   
ÿz.DistributedModel.fit_joblib.<locals>.<genexpr>c                 3   s&   | ]\}\}}ˆ ˆ|||ˆƒV  qd S rG   r   rW   rY   r   r   r[   i  s   
ÿc                 3   s,   | ]$\}\\}}}ˆ ˆ|||ˆ|ƒV  qd S rG   r   ©rX   r   rA   r%   r:   rY   r   r   r[   o  s   ÿc                 3   s,   | ]$\}\\}}}ˆ ˆ|||ˆ|ƒV  qd S rG   r   r\   rY   r   r   r[   v  s   ÿ)Zstatsmodels.tools.parallelrV   rE   r   rS   rT   )
r@   rP   r   rQ   rR   rV   ÚparZn_jobsr7   rU   r   rY   r   rO   D  s.    þ
þþ
þzDistributedModel.fit_joblib)NNNNNNNN)NrM   NN)N)N)Ú__name__Ú
__module__Ú__qualname__Ú__doc__rL   r   rN   rO   r   r   r   r   rF   m  s$   ?            ý
/    ÿ
: ÿ
0 ÿrF   c                       s(   e Zd ZdZ‡ fdd„Zdd„ Z‡  ZS )ÚDistributedResultsaT  
    Class to contain model results

    Parameters
    ----------
    model : class instance
        Class instance for model used for distributed data,
        this particular instance uses fake data and is really
        only to allow use of methods like predict.
    params : ndarray
        Parameter estimates from the fit model.
    c                    s   t t| ƒ ||¡ d S rG   )Úsuperrb   rL   )r@   rC   r
   ©Ú	__class__r   r   rL   ‹  s    zDistributedResults.__init__c                 O   s   | j j| j|f|ž|ŽS )aè  Calls self.model.predict for the provided exog.  See
        Results.predict.

        Parameters
        ----------
        exog : array_like NOT optional
            The values for which we want to predict, unlike standard
            predict this is NOT optional since the data in self.model
            is fake.
        *args :
            Some models can take additional arguments. See the
            predict method of the model for the details.
        **kwargs :
            Some models can take additional keywords arguments. See the
            predict method of the model for the details.

        Returns
        -------
            prediction : ndarray, pandas.Series or pandas.DataFrame
            See self.model.predict
        )rC   Úpredictr
   )r@   r%   ÚargsÚkwargsr   r   r   rf   Ž  s    zDistributedResults.predict)r^   r_   r`   ra   rL   rf   Ú__classcell__r   r   rd   r   rb   }  s   rb   )N)N)r   )NNN)r   )Zstatsmodels.base.elastic_netr   Z(statsmodels.stats.regularized_covariancer   r   r   Zstatsmodels.base.modelr   Z#statsmodels.regression.linear_modelr   Únumpyr   r   r   r   r#   r'   r1   r9   rE   rF   rb   r   r   r   r   Ú<module>   s(   E


+    ÿ
A
.ÿ
%  