U
    ÃmœdÆ  ã                   @   sŽ  d Z ddlZddlmZ ddlmZ ddlm	Z	 G dd„ dƒZ
edk�rŠdgZdek�rŠd	Zejejjed
fd�e edf¡f Zeje e d
¡d
d¡e d¡ddd… f jZe ddddgddddgddddgg¡Ze ddddgddddgddddgg¡Ze ee¡Zedejjejd� 7 Ze edddg¡Zedejjejd�  Ze
eeƒZee  ¡ ƒ edƒ ej!dddd� ee  ¡ ƒ dS )zQ
Created on Sun Nov 14 08:21:41 2010

Author: josef-pktd
License: BSD (3-clause)
é    N)Úpca)ÚLeaveOneOutc                   @   s<   e Zd ZdZdd„ Zddd„Zd	d
„ Zddd„Zdd„ ZdS )ÚFactorModelUnivariatea  

    Todo:
    check treatment of const, make it optional ?
        add hasconst (0 or 1), needed when selecting nfact+hasconst
    options are arguments in calc_factors, should be more public instead
    cross-validation is slow for large number of observations
    c                 C   s   t  |¡| _t  |¡| _d S )N)ÚnpÚasarrayÚendogÚexog)Úselfr   r   © r
   úb/home/sam/Atlas/atlas_env/lib/python3.8/site-packages/statsmodels/sandbox/datarich/factormodels.pyÚ__init__   s    zFactorModelUnivariate.__init__Nr   Tc                 C   sn   |dkr| j }n
t |¡}t||dd�\}}}}|| _|rRtj|dd�| _d| _n|| _d| _|| _	|| _
dS )zÜget factor decomposition of exogenous variables

        This uses principal component analysis to obtain the factors. The number
        of factors kept is the maximum that will be considered in the regression.
        Né   )ÚkeepdimÚ	normalizeT)Úprependr   )r   r   r   r   Zexog_reducedÚsmZadd_constantÚfactorsÚhasconstÚevalsÚevecs)r	   Úxr   ZaddconstZxredÚfactr   r   r
   r
   r   Úcalc_factors!   s    
z"FactorModelUnivariate.calc_factorsc                 C   s:   t | dƒs|  ¡  t | j| jd d …d |d …f ¡ ¡ S )NZfactors_wconstr   )Úhasattrr   r   ÚOLSr   r   Úfit)r	   Znfactr
   r
   r   Úfit_fixed_nfact8   s    
z%FactorModelUnivariate.fit_fixed_nfactc                 C   s’  t | dƒs|  ¡  | j}|dkr0| jjd | }|| dk rDtdƒ‚t|dƒ}| j}g }td|| ƒD ]Ä}| jdd…d|…f }t	 
||¡ ¡ }	|�s
|dkrªtt|ƒƒ}d}
|D ]T\}}t	 
|| ||dd…f ¡ ¡ }|
|| |j |j||dd…f ¡ d 7 }
q²ntj}
| ||	j|	j|	j|
g¡ qft |¡ | _}tjt |dd…dd…f d	¡t |dd…df d	¡t |dd…d
f d	¡f | _dS )aW  estimate the model and selection criteria for up to maxfact factors

        The selection criteria that are calculated are AIC, BIC, and R2_adj. and
        additionally cross-validation prediction error sum of squares if `skip_crossval`
        is false. Cross-validation is not used by default because it can be
        time consuming to calculate.

        By default the cross-validation method is Leave-one-out on the full dataset.
        A different cross-validation sample can be specified as an argument to
        cv_iter.

        Results are attached in `results_find_nfact`



        r   Nr   zFnothing to do, number of factors (incl. constant) should be at least 1é
   ç        ç       @é   r   éÿÿÿÿ)r   r   r   r   ÚshapeÚ
ValueErrorÚminr   Úranger   r   r   r   ÚlenÚmodelZpredictÚparamsr   ÚnanÚappendZaicZbicZrsquared_adjÚarrayÚresults_find_nfactZr_ZargminZargmaxÚ
best_nfact)r	   ÚmaxfactÚskip_crossvalÚcv_iterr   Úy0ÚresultsÚkr   ÚresZprederr2ZinidxZoutidxZres_l1or
   r
   r   Úfit_find_nfact=   s<    

	 ÿÿ
4ÿz$FactorModelUnivariate.fit_find_nfactc                 C   s¾   t | dƒs|  ¡  | j}d}|d7 }|ddt| jƒ  7 }ddlm} d d	¡}d
gdgd  }t|d�}|||d|d�}|d7 }|d7 }|d| 	¡  7 }|d7 }|d7 }|d7 }|d7 }|S )zÄprovides a summary for the selection of the number of factors

        Returns
        -------
        sumstr : str
            summary of the results for selecting the number of factors

        r,   Ú z,
Best result for k, by AIC, BIC, R2_adj, L1Oz
                   z%5d %4d %6d %5dr   )ÚSimpleTablezk, AIC, BIC, R2_adj, L1Oz, z%6dz%10.3fé   )Z	data_fmtsN)Ztxt_fmtz"
PCA regression on simulated data,z+
DGP: 2 factors and 4 explanatory variablesÚ
z)
Notes: k is number of components of PCA,z&
       constant is added additionallyz-
       k=0 means regression on constant onlyz?
       L1O: sum of squared prediction errors for leave-one-out)
r   r5   r,   Útupler-   Zstatsmodels.iolib.tabler7   ÚsplitÚdictÚ__str__)r	   r2   Zsumstrr7   ÚheadersZ	numformatZtxt_fmt1Ztablr
   r
   r   Úsummary_find_nfact   s&    	


z(FactorModelUnivariate.summary_find_nfact)Nr   T)NTN)	Ú__name__Ú
__module__Ú__qualname__Ú__doc__r   r   r   r5   r?   r
   r
   r
   r   r      s   

Br   Ú__main__r   iô  é   )Úsizer8   r!   g      ð?r   g      @r   gš™™™™™¹?g      ø?zwith cross validation - slowerF)r.   r/   r0   )"rC   Únumpyr   Zstatsmodels.apiÚapir   Zstatsmodels.sandbox.toolsr   Z#statsmodels.sandbox.tools.cross_valr   r   r@   ZexamplesZnobsZc_ÚrandomÚnormalZonesZf0ÚrepeatÚeyeZarangeÚTZf2xcoefr+   ÚdotZx0r"   Zytruer1   ÚmodÚprintr?   r5   r
   r
   r
   r   Ú<module>   s:    

&0

þ

þ
