U
    ½mœdÐ\  ã                   @   sì   d dl Z d dlmZmZ d dlZd dlmZ ddl	m
Z
mZmZ ddl	mZ ddlmZ ddlmZmZ ddlmZ dd	lmZ dd
lmZ ddlmZmZ ddlmZ ddlmZ ddlmZ G dd„ deee
ƒZG dd„ deee
ƒZ dS )é    N)ÚIntegralÚRealé   )ÚBaseEstimatorÚClassifierMixinÚRegressorMixin)ÚMultiOutputMixin)Úcheck_random_state)Ú
StrOptionsÚInterval)Ú_num_samples)Úcheck_array)Úcheck_consistent_length)Úcheck_is_fittedÚ_check_sample_weight)Ú_random_choice_csc©Ú_weighted_percentile)Úclass_distributionc                       sŽ   e Zd ZU dZedddddhƒgdgeedd	gd
œZee	d< dd	d	d
œdd„Z
ddd„Zdd„ Zdd„ Zdd„ Zdd„ Zd‡ fdd„	Z‡  ZS )ÚDummyClassifieraX  DummyClassifier makes predictions that ignore the input features.

    This classifier serves as a simple baseline to compare against other more
    complex classifiers.

    The specific behavior of the baseline is selected with the `strategy`
    parameter.

    All strategies make predictions that ignore the input feature values passed
    as the `X` argument to `fit` and `predict`. The predictions, however,
    typically depend on values observed in the `y` parameter passed to `fit`.

    Note that the "stratified" and "uniform" strategies lead to
    non-deterministic predictions that can be rendered deterministic by setting
    the `random_state` parameter if needed. The other strategies are naturally
    deterministic and, once fit, always return the same constant prediction
    for any value of `X`.

    Read more in the :ref:`User Guide <dummy_estimators>`.

    .. versionadded:: 0.13

    Parameters
    ----------
    strategy : {"most_frequent", "prior", "stratified", "uniform",             "constant"}, default="prior"
        Strategy to use to generate predictions.

        * "most_frequent": the `predict` method always returns the most
          frequent class label in the observed `y` argument passed to `fit`.
          The `predict_proba` method returns the matching one-hot encoded
          vector.
        * "prior": the `predict` method always returns the most frequent
          class label in the observed `y` argument passed to `fit` (like
          "most_frequent"). ``predict_proba`` always returns the empirical
          class distribution of `y` also known as the empirical class prior
          distribution.
        * "stratified": the `predict_proba` method randomly samples one-hot
          vectors from a multinomial distribution parametrized by the empirical
          class prior probabilities.
          The `predict` method returns the class label which got probability
          one in the one-hot vector of `predict_proba`.
          Each sampled row of both methods is therefore independent and
          identically distributed.
        * "uniform": generates predictions uniformly at random from the list
          of unique classes observed in `y`, i.e. each class has equal
          probability.
        * "constant": always predicts a constant label that is provided by
          the user. This is useful for metrics that evaluate a non-majority
          class.

          .. versionchanged:: 0.24
             The default value of `strategy` has changed to "prior" in version
             0.24.

    random_state : int, RandomState instance or None, default=None
        Controls the randomness to generate the predictions when
        ``strategy='stratified'`` or ``strategy='uniform'``.
        Pass an int for reproducible output across multiple function calls.
        See :term:`Glossary <random_state>`.

    constant : int or str or array-like of shape (n_outputs,), default=None
        The explicit constant as predicted by the "constant" strategy. This
        parameter is useful only for the "constant" strategy.

    Attributes
    ----------
    classes_ : ndarray of shape (n_classes,) or list of such arrays
        Unique class labels observed in `y`. For multi-output classification
        problems, this attribute is a list of arrays as each output has an
        independent set of possible classes.

    n_classes_ : int or list of int
        Number of label for each output.

    class_prior_ : ndarray of shape (n_classes,) or list of such arrays
        Frequency of each class observed in `y`. For multioutput classification
        problems, this is computed independently for each output.

    n_outputs_ : int
        Number of outputs.

    sparse_output_ : bool
        True if the array returned from predict is to be in sparse CSC format.
        Is automatically set to True if the input `y` is passed in sparse
        format.

    See Also
    --------
    DummyRegressor : Regressor that makes predictions using simple rules.

    Examples
    --------
    >>> import numpy as np
    >>> from sklearn.dummy import DummyClassifier
    >>> X = np.array([-1, 1, 1, 1])
    >>> y = np.array([0, 1, 1, 1])
    >>> dummy_clf = DummyClassifier(strategy="most_frequent")
    >>> dummy_clf.fit(X, y)
    DummyClassifier(strategy='most_frequent')
    >>> dummy_clf.predict(X)
    array([1, 1, 1, 1])
    >>> dummy_clf.score(X, y)
    0.75
    Úmost_frequentÚpriorÚ
stratifiedÚuniformÚconstantÚrandom_stateú
array-likeN©Ústrategyr   r   Ú_parameter_constraintsc                C   s   || _ || _|| _d S ©Nr   )Úselfr   r   r   © r"   úF/home/sam/Atlas/atlas_env/lib/python3.8/site-packages/sklearn/dummy.pyÚ__init__Œ   s    zDummyClassifier.__init__c                    s”  |   ¡  | j| _| jdkr8t |¡r8| ¡ }t dt¡ t |¡| _	| j	s^t
 |¡}t
 |¡}|jdkrtt
 |d¡}|jd | _t||ƒ |dk	rœt||ƒ}| jdkrì| jdkrºtdƒ‚n2t
 t
 | j¡d¡‰ ˆ jd | jkrìtd	| j ƒ‚t||ƒ\| _| _| _| jdk�r`t| jƒD ]F‰t‡ ‡fd
d„| jˆ D ƒƒ�sd | jt| jˆ ƒ¡}t|ƒ‚�q| jdk�r�| jd | _| jd | _| jd | _| S )aÆ  Fit the baseline classifier.

        Parameters
        ----------
        X : array-like of shape (n_samples, n_features)
            Training data.

        y : array-like of shape (n_samples,) or (n_samples, n_outputs)
            Target values.

        sample_weight : array-like of shape (n_samples,), default=None
            Sample weights.

        Returns
        -------
        self : object
            Returns the instance itself.
        r   zªA local copy of the target data has been converted to a numpy array. Predicting on sparse target data with the uniform strategy would not save memory and would be slower.r   ©éÿÿÿÿr   Nr   úMConstant target value has to be specified when the constant strategy is used.r   ú0Constant target value should have shape (%d, 1).c                 3   s   | ]}ˆ ˆ d  |kV  qdS )r   Nr"   ©Ú.0Úc©r   Úkr"   r#   Ú	<genexpr>Ö   s     z&DummyClassifier.fit.<locals>.<genexpr>zrThe constant target value must be present in the training data. You provided constant={}. Possible values are: {}.)Ú_validate_paramsr   Ú	_strategyÚspÚissparseZtoarrayÚwarningsÚwarnÚUserWarningÚsparse_output_ÚnpZasarrayZ
atleast_1dÚndimÚreshapeÚshapeÚ
n_outputs_r   r   r   Ú
ValueErrorr   Úclasses_Ú
n_classes_Úclass_prior_ÚrangeÚanyÚformatÚlist)r!   ÚXÚyÚsample_weightÚerr_msgr"   r,   r#   Úfit‘   s`    û






ÿÿÿ ÿ  ýÿzDummyClassifier.fitc                    s¾  t | ƒ t|ƒ‰t| jƒ‰| j‰| j‰| j‰ | j}| jdkrTˆg‰ˆg‰ˆ g‰ |g}| j	dkrx|  
|¡‰| jdkrxˆg‰| jrêd}| j	dkrœdd„ ˆ D ƒ‰n<| j	dkr¬ˆ }n,| j	dkrÀtdƒ‚n| j	d	krØd
d„ |D ƒ‰tˆˆ|| jƒ}nÐ| j	dk�rt ‡ ‡fdd„t| jƒD ƒˆdg¡}n†| j	dk�rNt ‡‡fdd„t| jƒD ƒ¡j}nV| j	dk�r†‡‡‡‡fdd„t| jƒD ƒ}t |¡j}n| j	d	k�r¤t | jˆdf¡}| jdk�rºt |¡}|S )a;  Perform classification on test vectors X.

        Parameters
        ----------
        X : array-like of shape (n_samples, n_features)
            Test data.

        Returns
        -------
        y : array-like of shape (n_samples,) or (n_samples, n_outputs)
            Predicted target values for X.
        r   r   N)r   r   c                 S   s   g | ]}t  | ¡ g¡‘qS r"   )r7   ÚarrayÚargmax)r*   Úcpr"   r"   r#   Ú
<listcomp>  s     z+DummyClassifier.predict.<locals>.<listcomp>r   zCSparse target prediction is not supported with the uniform strategyr   c                 S   s   g | ]}t  |g¡‘qS r"   )r7   rI   r)   r"   r"   r#   rL     s     c                    s    g | ]}ˆ| ˆ |   ¡  ‘qS r"   ©rJ   ©r*   r-   )r?   r=   r"   r#   rL   "  s   ÿc                    s$   g | ]}ˆ | ˆ| j d d� ‘qS )r   ©ÚaxisrM   rN   )r=   Úprobar"   r#   rL   +  s   ÿc                    s&   g | ]}ˆ | ˆj ˆ| ˆd � ‘qS )©Úsize)ÚrandintrN   )r=   r>   Ú	n_samplesÚrsr"   r#   rL   2  s   ÿ)r   r   r	   r   r>   r=   r?   r   r;   r0   Úpredict_probar6   r<   r   r7   Ztiler@   ZvstackÚTÚravel)r!   rD   r   Z
class_probrE   Úretr"   )r?   r=   r>   rU   rQ   rV   r#   Úpredicté   sh    







ÿ
þûþÿþ
zDummyClassifier.predictc                 C   s–  t | ƒ t|ƒ}t| jƒ}| j}| j}| j}| j}| jdkrT|g}|g}|g}|g}g }t	| jƒD �]}	| j
dkr¨||	  ¡ }
tj|||	 ftjd�}d|dd…|
f< nÊ| j
dkrÊt |df¡||	  }n¨| j
dkrö|jd||	 |d�}| tj¡}n|| j
d	k�r(tj|||	 ftjd�}|||	  }nJ| j
d
k�rrt ||	 ||	 k¡}
tj|||	 ftjd�}d|dd…|
f< | |¡ qb| jdk�r’|d }|S )aÊ  
        Return probability estimates for the test vectors X.

        Parameters
        ----------
        X : array-like of shape (n_samples, n_features)
            Test data.

        Returns
        -------
        P : ndarray of shape (n_samples, n_classes) or list of such arrays
            Returns the probability of the sample for each class in
            the model, where classes are ordered arithmetically, for each
            output.
        r   r   ©Údtypeç      ð?Nr   r   rR   r   r   r   )r   r   r	   r   r>   r=   r?   r   r;   r@   r0   rJ   r7   ÚzerosZfloat64ZonesZmultinomialZastypeÚwhereÚappend)r!   rD   rU   rV   r>   r=   r?   r   ÚPr-   ÚindÚoutr"   r"   r#   rW   @  sD    




zDummyClassifier.predict_probac                 C   s0   |   |¡}| jdkrt |¡S dd„ |D ƒS dS )aÚ  
        Return log probability estimates for the test vectors X.

        Parameters
        ----------
        X : {array-like, object with finite length or shape}
            Training data.

        Returns
        -------
        P : ndarray of shape (n_samples, n_classes) or list of such arrays
            Returns the log probability of the sample for each class in
            the model, where classes are ordered arithmetically for each
            output.
        r   c                 S   s   g | ]}t  |¡‘qS r"   )r7   Úlog)r*   Úpr"   r"   r#   rL   “  s     z5DummyClassifier.predict_log_proba.<locals>.<listcomp>N)rW   r;   r7   re   )r!   rD   rQ   r"   r"   r#   Úpredict_log_proba  s    


z!DummyClassifier.predict_log_probac                 C   s   dddddœdœS )NTzfails for the predict method)Zcheck_methods_subset_invarianceZ%check_methods_sample_order_invariance)Ú
poor_scoreÚno_validationZ_xfail_checksr"   ©r!   r"   r"   r#   Ú
_more_tags•  s    þýzDummyClassifier._more_tagsc                    s,   |dkrt jt|ƒdfd�}tƒ  |||¡S )ak  Return the mean accuracy on the given test data and labels.

        In multi-label classification, this is the subset accuracy
        which is a harsh metric since you require for each sample that
        each label set be correctly predicted.

        Parameters
        ----------
        X : None or array-like of shape (n_samples, n_features)
            Test samples. Passing None as test samples gives the same result
            as passing real test samples, since DummyClassifier
            operates independently of the sampled observations.

        y : array-like of shape (n_samples,) or (n_samples, n_outputs)
            True labels for X.

        sample_weight : array-like of shape (n_samples,), default=None
            Sample weights.

        Returns
        -------
        score : float
            Mean accuracy of self.predict(X) w.r.t. y.
        Nr   ©r:   ©r7   r_   ÚlenÚsuperÚscore©r!   rD   rE   rF   ©Ú	__class__r"   r#   rp   Ÿ  s    zDummyClassifier.score)N)N)Ú__name__Ú
__module__Ú__qualname__Ú__doc__r
   r   Ústrr   ÚdictÚ__annotations__r$   rH   r[   rW   rg   rk   rp   Ú__classcell__r"   r"   rr   r#   r      s   
lÿ
û
XW?
r   c                       s–   e Zd ZU dZeddddhƒgeedddd	�d
geed
d
dd	�dd
gdœZee	d< dd
d
dœdd„Z
ddd„Zddd„Zdd„ Zd‡ fdd„	Z‡  ZS )ÚDummyRegressoraš  Regressor that makes predictions using simple rules.

    This regressor is useful as a simple baseline to compare with other
    (real) regressors. Do not use it for real problems.

    Read more in the :ref:`User Guide <dummy_estimators>`.

    .. versionadded:: 0.13

    Parameters
    ----------
    strategy : {"mean", "median", "quantile", "constant"}, default="mean"
        Strategy to use to generate predictions.

        * "mean": always predicts the mean of the training set
        * "median": always predicts the median of the training set
        * "quantile": always predicts a specified quantile of the training set,
          provided with the quantile parameter.
        * "constant": always predicts a constant value that is provided by
          the user.

    constant : int or float or array-like of shape (n_outputs,), default=None
        The explicit constant as predicted by the "constant" strategy. This
        parameter is useful only for the "constant" strategy.

    quantile : float in [0.0, 1.0], default=None
        The quantile to predict using the "quantile" strategy. A quantile of
        0.5 corresponds to the median, while 0.0 to the minimum and 1.0 to the
        maximum.

    Attributes
    ----------
    constant_ : ndarray of shape (1, n_outputs)
        Mean or median or quantile of the training targets or constant value
        given by the user.

    n_outputs_ : int
        Number of outputs.

    See Also
    --------
    DummyClassifier: Classifier that makes predictions using simple rules.

    Examples
    --------
    >>> import numpy as np
    >>> from sklearn.dummy import DummyRegressor
    >>> X = np.array([1.0, 2.0, 3.0, 4.0])
    >>> y = np.array([2.0, 3.0, 5.0, 10.0])
    >>> dummy_regr = DummyRegressor(strategy="mean")
    >>> dummy_regr.fit(X, y)
    DummyRegressor()
    >>> dummy_regr.predict(X)
    array([5., 5., 5., 5.])
    >>> dummy_regr.score(X, y)
    0.0
    ÚmeanÚmedianÚquantiler   g        r^   Zboth)ÚclosedNZneitherr   )r   r   r   r   ©r   r   r   c                C   s   || _ || _|| _d S r    r�   )r!   r   r   r   r"   r"   r#   r$     s    zDummyRegressor.__init__c                    s¶  |   ¡  tˆddd�‰tˆƒdkr*tdƒ‚ˆjdkr@t ˆd¡‰ˆjd | _t	|ˆˆƒ ˆdk	rjt
ˆ|ƒ‰| jd	krŠtjˆdˆd
�| _�n| jdkrÌˆdkr®tjˆdd�| _n‡‡fdd„t| jƒD ƒ| _nÖ| jdk�r2| jdkrêtdƒ‚| jd ‰ ˆdk�rtjˆdˆ d�| _n‡ ‡‡fdd„t| jƒD ƒ| _np| jdk�r¢| jdk�rRtdƒ‚t| jdddgddd�| _| jdk�r¢| jjd ˆjd k�r¢tdˆjd  ƒ‚t | jd¡| _| S )a¸  Fit the random regressor.

        Parameters
        ----------
        X : array-like of shape (n_samples, n_features)
            Training data.

        y : array-like of shape (n_samples,) or (n_samples, n_outputs)
            Target values.

        sample_weight : array-like of shape (n_samples,), default=None
            Sample weights.

        Returns
        -------
        self : object
            Fitted estimator.
        FrE   )Ú	ensure_2dZ
input_namer   zy must not be empty.r   r%   Nr}   )rP   Úweightsr~   rO   c                    s&   g | ]}t ˆd d …|f ˆ dd�‘qS )Ng      I@©Ú
percentiler   rN   )rF   rE   r"   r#   rL   0  s   ÿz&DummyRegressor.fit.<locals>.<listcomp>r   z^When using `strategy='quantile', you have to specify the desired quantile in the range [0, 1].g      Y@)rP   Úqc                    s&   g | ]}t ˆd d …|f ˆˆ d�‘qS )Nr„   r   rN   ©r…   rF   rE   r"   r#   rL   ?  s   ÿr   r'   ZcsrZcscZcoo)Zaccept_sparser‚   Zensure_min_samplesr(   )r   r&   )r/   r   rn   r<   r8   r7   r9   r:   r;   r   r   r   ZaverageÚ	constant_r~   r@   r   r…   r   Ú	TypeErrorrq   r"   r‡   r#   rH     s\    



þ

ÿ

þ
ÿü$ÿzDummyRegressor.fitFc                 C   sp   t | ƒ t|ƒ}tj|| jf| jt | j¡jd�}t || jf¡}| jdkr`t 	|¡}t 	|¡}|rl||fS |S )a’  Perform classification on test vectors X.

        Parameters
        ----------
        X : array-like of shape (n_samples, n_features)
            Test data.

        return_std : bool, default=False
            Whether to return the standard deviation of posterior prediction.
            All zeros in this case.

            .. versionadded:: 0.20

        Returns
        -------
        y : array-like of shape (n_samples,) or (n_samples, n_outputs)
            Predicted target values for X.

        y_std : array-like of shape (n_samples,) or (n_samples, n_outputs)
            Standard deviation of predictive distribution of query points.
        r\   r   )
r   r   r7   Úfullr;   rˆ   rI   r]   r_   rY   )r!   rD   Z
return_stdrU   rE   Zy_stdr"   r"   r#   r[   Z  s    ý


zDummyRegressor.predictc                 C   s
   dddœS )NT)rh   ri   r"   rj   r"   r"   r#   rk   €  s    zDummyRegressor._more_tagsc                    s,   |dkrt jt|ƒdfd�}tƒ  |||¡S )aŽ  Return the coefficient of determination R^2 of the prediction.

        The coefficient R^2 is defined as `(1 - u/v)`, where `u` is the
        residual sum of squares `((y_true - y_pred) ** 2).sum()` and `v` is the
        total sum of squares `((y_true - y_true.mean()) ** 2).sum()`. The best
        possible score is 1.0 and it can be negative (because the model can be
        arbitrarily worse). A constant model that always predicts the expected
        value of y, disregarding the input features, would get a R^2 score of
        0.0.

        Parameters
        ----------
        X : None or array-like of shape (n_samples, n_features)
            Test samples. Passing None as test samples gives the same result
            as passing real test samples, since `DummyRegressor`
            operates independently of the sampled observations.

        y : array-like of shape (n_samples,) or (n_samples, n_outputs)
            True values for X.

        sample_weight : array-like of shape (n_samples,), default=None
            Sample weights.

        Returns
        -------
        score : float
            R^2 of `self.predict(X)` w.r.t. y.
        Nr   rl   rm   rq   rr   r"   r#   rp   ƒ  s    zDummyRegressor.score)N)F)N)rt   ru   rv   rw   r
   r   r   r   ry   rz   r$   rH   r[   rk   rp   r{   r"   r"   rr   r#   r|   ½  s   
;ýý

S
&r|   )!r3   Únumbersr   r   Únumpyr7   Zscipy.sparseÚsparser1   Úbaser   r   r   r   Úutilsr	   Zutils._param_validationr
   r   Zutils.validationr   r   r   r   r   Zutils.randomr   Zutils.statsr   Zutils.multiclassr   r   r|   r"   r"   r"   r#   Ú<module>   s&      '