U
    ÅmœdÉŽ  ã                   @   sÚ  d Z ddlmZ ddlmZ ddlZddlmZmZm	Z	m
Z
mZmZ ddlZddlZddlZddlmZmZmZmZ ddlmZmZ ddlmZ dd	lmZ d
dlmZ  d
dl!m"Z# d
dl$m%Z%m&Z&m'Z'm(Z(m)Z) d
dl*m+Z+ d
dl,m-Z-m.Z. ddl/m0Z0 ddl$m1Z1 zddl2m3Z4 W n e5k
�r.   dZ4Y nX ddl6m7Z7 dVeee8 ee8 ee8 ee8 e9e9ee	ej:ej:f  dœdd„Z;dWeee8 ee8 ee8 ee8 e9e9eede	ej:ej:f f dœdd„Z<edddddddœeeej:ef ee e9e9ee8 ee= ee= dœdd„ƒZ>e> ?e¡ddd œee e9d œd!d"„ƒZ@e> ?ej:¡ddd œee e9d œd#d$„ƒZAe> ?e¡dddddddœee e9e9ee8 ee= ee= ee d%œd&d'„ƒZBdXee9e9ee8 ee d(œd)d*„ZCdYeeej:ef eeD eej: e=e9ee+d- ee= f ee+d.  e8ee d/œ	d0d1„ZEdZeee=ee= f ee8 e9ee d2œd3d4„ZFd5d6„ ZGed[eeeej:f e9eeD e9ee= ee= d7œd8d9„ƒZHeH ?ej:¡ddddd:œe9eeD e9e9d:œd;d<„ƒZIeH ?e¡ddddd:œe9eeD e9e9d:œd=d>„ƒZJeH ?e¡dddddd?œee9eeD e9ee= ee= ee d@œdAdB„ƒZKd\eeej:ef eeD ee8 e(e9ee dCœdDdE„ZLe&dFdGiƒd]ddddHœeeee8e
e8 f  ee8 e(e9e9ee dIœdJdK„ƒZMdLdM„ ZNdNdO„ ZOejPddP�d^ej:e8e(e9e9dQœdRdS„ƒZQd_dTdU„ZRdS )`zdSimple Preprocessing Functions

Compositions of these functions are found in sc.preprocess.recipes.
é    )Úsingledispatch)ÚNumberN)ÚUnionÚOptionalÚTupleÚ
CollectionÚSequenceÚIterable)ÚissparseÚisspmatrix_csrÚ
csr_matrixÚspmatrix)ÚsparsefuncsÚcheck_array)Úis_categorical_dtype)ÚAnnDataé   )Úlogging)Úsettings)Úsanitize_anndataÚdeprecated_arg_namesÚview_to_actualÚ	AnyRandomÚ_check_array_function_arguments)ÚLiteral)Ú_get_obs_repÚ_set_obs_repé   )Úmaterialize_as_ndarray)Ú_get_mean_var)Úfilter_genes_dispersionTF)ÚdataÚ
min_countsÚ	min_genesÚ
max_countsÚ	max_genesÚinplaceÚcopyÚreturnc                 C   sâ  |rt  d¡ tdd„ ||||fD ƒƒ}|dkr8tdƒ‚t| tƒr´|rN|  ¡ n| }tt|j	||||ƒƒ\}	}
|sx|	|
fS |dkr”|dkr”|
|j
d< n
|
|j
d< | |	¡ |r°|S dS | }|dkrÄ|n|}|dkrÔ|n|}tj|dkrð|dkrð|n|d	kdd
�}t|ƒ�r|j}|dk	�r ||k}	|dk	�r2||k}	t |	 ¡}|d	k�rÚd|› d�}|dk	�sh|dk	�r’|d7 }||dk�r†|› d�n|› d�7 }|dk	�s¦|dk	�rÐ|d7 }||dk�rÄ|› d�n|› d�7 }t  |¡ |	|fS )u£      Filter cell outliers based on counts and numbers of genes expressed.

    For instance, only keep cells with at least `min_counts` counts or
    `min_genes` genes expressed. This is to filter measurement outliers,
    i.e. â€œunreliableâ€� observations.

    Only provide one of the optional parameters `min_counts`, `min_genes`,
    `max_counts`, `max_genes` per call.

    Parameters
    ----------
    data
        The (annotated) data matrix of shape `n_obs` Ã— `n_vars`.
        Rows correspond to cells and columns to genes.
    min_counts
        Minimum number of counts required for a cell to pass filtering.
    min_genes
        Minimum number of genes expressed required for a cell to pass filtering.
    max_counts
        Maximum number of counts required for a cell to pass filtering.
    max_genes
        Maximum number of genes expressed required for a cell to pass filtering.
    inplace
        Perform computation inplace or return result.

    Returns
    -------
    Depending on `inplace`, returns the following arrays or directly subsets
    and annotates the data matrix:

    cells_subset
        Boolean index mask that does filtering. `True` means that the
        cell is kept. `False` means the cell is removed.
    number_per_cell
        Depending on what was tresholded (`counts` or `genes`),
        the array stores `n_counts` or `n_cells` per gene.

    Examples
    --------
    >>> import scanpy as sc
    >>> adata = sc.datasets.krumsiek11()
    >>> adata.n_obs
    640
    >>> adata.var_names
    ['Gata2' 'Gata1' 'Fog1' 'EKLF' 'Fli1' 'SCL' 'Cebpa'
     'Pu.1' 'cJun' 'EgrNab' 'Gfi1']
    >>> # add some true zeros
    >>> adata.X[adata.X < 0.3] = 0
    >>> # simply compute the number of genes per cell
    >>> sc.pp.filter_cells(adata, min_genes=0)
    >>> adata.n_obs
    640
    >>> adata.obs['n_genes'].min()
    1
    >>> # filter manually
    >>> adata_copy = adata[adata.obs['n_genes'] >= 3]
    >>> adata_copy.obs['n_genes'].min()
    >>> adata.n_obs
    554
    >>> adata.obs['n_genes'].min()
    3
    >>> # actually do some filtering
    >>> sc.pp.filter_cells(adata, min_genes=3)
    >>> adata.n_obs
    554
    >>> adata.obs['n_genes'].min()
    3
    ú,`copy` is deprecated, use `inplace` instead.c                 s   s   | ]}|d k	V  qd S ©N© ©Ú.0Úoptionr+   r+   úU/home/sam/Atlas/atlas_env/lib/python3.8/site-packages/scanpy/preprocessing/_simple.pyÚ	<genexpr>z   s    zfilter_cells.<locals>.<genexpr>r   zjOnly provide one of the optional parameters `min_counts`, `min_genes`, `max_counts`, `max_genes` per call.NÚn_countsZn_genesr   ©Zaxisúfiltered out z cells that have z
less than z genes expressedú countsz
more than )ÚloggÚwarningÚsumÚ
ValueErrorÚ
isinstancer   r'   r   Úfilter_cellsÚXÚobsÚ_inplace_subset_obsÚnpr
   ÚA1Úinfo)r!   r"   r#   r$   r%   r&   r'   Ún_given_optionsÚadataÚcell_subsetÚnumberr;   Ú
min_numberÚ
max_numberZnumber_per_cellÚsÚmsgr+   r+   r/   r:   *   sj    N

ÿÿ
ÿ

 ÿ



ÿýÿý
r:   )r!   r"   Ú	min_cellsr$   Ú	max_cellsr&   r'   r(   c                 C   sä  |rt  d¡ tdd„ ||||fD ƒƒ}|dkr8tdƒ‚t| tƒr¶|rN|  ¡ n| }tt|j	||||d�ƒ\}	}
|sz|	|
fS |dkr–|dkr–|
|j
d< n
|
|j
d	< | |	¡ |r²|S dS | }|dkrÆ|n|}|dkrÖ|n|}tj|dkrò|dkrò|n|d
kd
d�}t|ƒ�r|j}|dk	�r"||k}	|dk	�r4||k}	t |	 ¡}|d
k�rÜd|› d�}|dk	�sj|dk	�r”|d7 }||dk�rˆ|› d�n|› d�7 }|dk	�s¨|dk	�rÒ|d7 }||dk�rÆ|› d�n|› d�7 }t  |¡ |	|fS )ua      Filter genes based on number of cells or counts.

    Keep genes that have at least `min_counts` counts or are expressed in at
    least `min_cells` cells or have at most `max_counts` counts or are expressed
    in at most `max_cells` cells.

    Only provide one of the optional parameters `min_counts`, `min_cells`,
    `max_counts`, `max_cells` per call.

    Parameters
    ----------
    data
        An annotated data matrix of shape `n_obs` Ã— `n_vars`. Rows correspond
        to cells and columns to genes.
    min_counts
        Minimum number of counts required for a gene to pass filtering.
    min_cells
        Minimum number of cells expressed required for a gene to pass filtering.
    max_counts
        Maximum number of counts required for a gene to pass filtering.
    max_cells
        Maximum number of cells expressed required for a gene to pass filtering.
    inplace
        Perform computation inplace or return result.

    Returns
    -------
    Depending on `inplace`, returns the following arrays or directly subsets
    and annotates the data matrix

    gene_subset
        Boolean index mask that does filtering. `True` means that the
        gene is kept. `False` means the gene is removed.
    number_per_gene
        Depending on what was tresholded (`counts` or `cells`), the array stores
        `n_counts` or `n_cells` per gene.
    r)   c                 s   s   | ]}|d k	V  qd S r*   r+   r,   r+   r+   r/   r0   â   s    zfilter_genes.<locals>.<genexpr>r   zjOnly provide one of the optional parameters `min_counts`, `min_cells`, `max_counts`, `max_cells` per call.)rI   r"   rJ   r$   Nr1   Zn_cellsr   r2   r3   z genes that are detected zin less than z cellsr4   zin more than )r5   r6   r7   r8   r9   r   r'   r   Úfilter_genesr;   ÚvarZ_inplace_subset_varr>   r
   r?   r@   )r!   r"   rI   r$   rJ   r&   r'   rA   rB   Zgene_subsetrD   r;   rE   rF   Znumber_per_generG   rH   r+   r+   r/   rK   ±   sn    /

ÿÿ
ûÿ	

 ÿ



ÿÿ
rK   )Úbaser'   ÚchunkedÚ
chunk_sizeÚlayerÚobsm©r;   rM   r'   rN   rO   rP   rQ   c                C   s   t ||||d� t| ||d�S )uj      Logarithmize the data matrix.

    Computes :math:`X = \log(X + 1)`,
    where :math:`log` denotes the natural logarithm unless a different base is given.

    Parameters
    ----------
    X
        The (annotated) data matrix of shape `n_obs` Ã— `n_vars`.
        Rows correspond to cells and columns to genes.
    base
        Base of the logarithm. Natural logarithm is used by default.
    copy
        If an :class:`~anndata.AnnData` is passed, determines whether a copy
        is returned.
    chunked
        Process the data matrix in chunks, which will save memory.
        Applies only to :class:`~anndata.AnnData`.
    chunk_size
        `n_obs` of the chunks to process the data in.
    layer
        Entry of layers to tranform.
    obsm
        Entry of obsm to transform.

    Returns
    -------
    Returns or updates `data`, depending on `copy`.
    )rN   rO   rP   rQ   ©r'   rM   )r   Úlog1p_arrayrR   r+   r+   r/   Úlog1p  s    )   ÿrU   ©rM   r'   c                C   s.   t | dtjtjf|d�} t| jd|d�| _| S )N)ZcsrZcsc)Zaccept_sparseÚdtyper'   FrS   )r   r>   Zfloat64Úfloat32rU   r!   ©r;   rM   r'   r+   r+   r/   Úlog1p_sparseL  s      
 ÿrZ   c                C   s‚   |r*t  | jt j¡s |  t¡} qR|  ¡ } n(t  | jt j¡sRt  | jt¡sR|  t¡} t j| | d� |d k	r~t j	| t  
|¡| d� | S )N)Úout)r>   Ú
issubdtyperW   ZfloatingÚastypeÚfloatr'   ÚcomplexrU   ÚdivideÚlogrY   r+   r+   r/   rT   U  s    

rT   )rM   r'   rN   rO   rP   rQ   r(   c                C   sÀ   d|   ¡ krt d¡ |r"|  ¡ n| } t| ƒ |rz|d k	sB|d k	rJtdƒ‚|  |¡D ]"\}}}	t||dd�| j||	…< qTn,t	| ||d�}
t|
d|d�}
t
| |
||d� d|i| jd< |r¼| S d S )	NrU   z,adata.X seems to be already log-transformed.zFCurrently cannot perform chunked operations on arrays not stored in X.FrV   ©rP   rQ   rS   rM   )Zuns_keysr5   r6   r'   r   ÚNotImplementedErrorÚ	chunked_XrU   r;   r   r   Zuns)rB   rM   r'   rN   rO   rP   rQ   ÚchunkÚstartÚendr;   r+   r+   r/   Úlog1p_anndataf  s"    
ÿrh   )r!   r'   rN   rO   r(   c           	      C   s‚   t | tƒr`|r|  ¡ n| }|rH| |¡D ]\}}}t|ƒ|j||…< q(nt| jƒ|_|r\|S dS | }t|ƒsvt |¡S | ¡ S dS )up      Square root the data matrix.

    Computes :math:`X = \sqrt(X)`.

    Parameters
    ----------
    data
        The (annotated) data matrix of shape `n_obs` Ã— `n_vars`.
        Rows correspond to cells and columns to genes.
    copy
        If an :class:`~anndata.AnnData` object is passed,
        determines whether a copy is returned.
    chunked
        Process the data matrix in chunks, which will save memory.
        Applies only to :class:`~anndata.AnnData`.
    chunk_size
        `n_obs` of the chunks to process the data in.

    Returns
    -------
    Returns or updates `data`, depending on `copy`.
    N)r9   r   r'   rd   Úsqrtr;   r
   r>   )	r!   r'   rN   rO   rB   re   rf   rg   r;   r+   r+   r/   ri   ˆ  s    

ri   r1   r+   Úall)Úafterr;   )	r!   Úcounts_per_cell_afterÚcounts_per_cellÚkey_n_countsr'   ÚlayersÚuse_repr"   r(   c              	   C   sø  t | tƒ�r$t d¡}|r"|  ¡ n| }	|dkr`tt|	j|d�ƒ\}
}||	j|< |	 	|
¡ ||
 }t
|	j||ƒ |dkr€|	j ¡ n|}|dkr’|}n.|dkrªt ||
 ¡}n|dkr¸d}ntdƒ‚|D ]:}t|	j| |d�\}}t
|	j| ||dd	�}||	j|< qÄtjd
|›d�|d� |�r |	S dS |�r2|  ¡ n| }|dk�rn|�sNtdƒ‚t||d�\}
}||
 }||
 }|dk�r‚t |¡}t ¡ �Z t d¡ ||dk7 }|| }t|ƒ�sÐ|t|dd…tjf ƒ }nt |d| ¡ W 5 Q R X |�rô|S dS )u½      Normalize total counts per cell.

    .. warning::
        .. deprecated:: 1.3.7
            Use :func:`~scanpy.pp.normalize_total` instead.
            The new function is equivalent to the present
            function, except that

            * the new function doesn't filter cells based on `min_counts`,
              use :func:`~scanpy.pp.filter_cells` if filtering is needed.
            * some arguments were renamed
            * `copy` is replaced by `inplace`

    Normalize each cell by total counts over all genes, so that every cell has
    the same total count after normalization.

    Similar functions are used, for example, by Seurat [Satija15]_, Cell Ranger
    [Zheng17]_ or SPRING [Weinreb17]_.

    Parameters
    ----------
    data
        The (annotated) data matrix of shape `n_obs` Ã— `n_vars`. Rows correspond
        to cells and columns to genes.
    counts_per_cell_after
        If `None`, after normalization, each cell has a total count equal
        to the median of the *counts_per_cell* before normalization.
    counts_per_cell
        Precomputed counts per cell.
    key_n_counts
        Name of the field in `adata.obs` where the total counts per cell are
        stored.
    copy
        If an :class:`~anndata.AnnData` is passed, determines whether a copy
        is returned.
    min_counts
        Cells with counts less than `min_counts` are filtered out during
        normalization.

    Returns
    -------
    Returns or updates `adata` with normalized version of the original
    `adata.X`, depending on `copy`.

    Examples
    --------
    >>> import scanpy as sc
    >>> adata = AnnData(np.array([[1, 0], [3, 0], [5, 6]]))
    >>> print(adata.X.sum(axis=1))
    [  1.   3.  11.]
    >>> sc.pp.normalize_per_cell(adata)
    >>> print(adata.obs)
    >>> print(adata.X.sum(axis=1))
       n_counts
    0       1.0
    1       3.0
    2      11.0
    [ 3.  3.  3.]
    >>> sc.pp.normalize_per_cell(
    >>>     adata, counts_per_cell_after=1,
    >>>     key_n_counts='n_counts2',
    >>> )
    >>> print(adata.obs)
    >>> print(adata.X.sum(axis=1))
       n_counts  n_counts2
    0       1.0        3.0
    1       3.0        3.0
    2      11.0        3.0
    [ 1.  1.  1.]
    z#normalizing by total count per cellN)r"   rj   rk   r;   z&use_rep should be "after", "X" or NoneT©r'   z>    finished ({time_passed}): normalized adata.X and added    z2, counts per cell before normalization (adata.obs)©ÚtimezCan only be run with copy=TrueÚignorer   r   )r9   r   r5   r@   r'   r   r:   r;   r<   r=   Únormalize_per_cellro   Úkeysr>   Zmedianr8   ÚwarningsÚcatch_warningsÚsimplefilterr
   Znewaxisr   Zinplace_row_scale)r!   rl   rm   rn   r'   ro   rp   r"   rf   rB   rC   rk   rP   ZsubsetÚcountsÚtempr;   r+   r+   r/   ru   ´  sZ    Q
ÿ


ý





ru   )rB   rv   Ún_jobsr'   r(   c                    sp  t  d|› �¡}t| jƒr$t  d¡ |r0|  ¡ n| } t| ƒ | jrP|  |  ¡ ¡ t|t	ƒr`|g}t| jƒrv| j 
¡ | _|dkr„tjn|}d}|d |  ¡ k�r@t| j|d  ƒ�r@t|ƒdkrÆtdƒ‚t  d¡ tj| jjd	d
�}| j|d  jjD ]D}|| j|d  kj}t| jjƒD ]\}	}
|
|  ¡ |||	f< �qqôd}n*|�rR| j| }n
| j ¡ }| ddd¡ t td| jjd ƒ| ¡ t ¡}t | jjd | ¡ t ¡}g }tj!| j|dd�}|�rÔtj!||dd�}t|ƒD ]2\}}|�rô|| }n|}| "t#|||fƒ¡ �qÜddl$m%}m&‰  ||d�‡ fdd„|D ƒƒ}t '|¡j | jj(¡| _t jd|d� |�rl| S dS )aÙ      Regress out (mostly) unwanted sources of variation.

    Uses simple linear regression. This is inspired by Seurat's `regressOut`
    function in R [Satija15]. Note that this function tends to overcorrect
    in certain circumstances as described in :issue:`526`.

    Parameters
    ----------
    adata
        The annotated data matrix.
    keys
        Keys for observation annotation on which to regress on.
    n_jobs
        Number of jobs for parallel computation.
        `None` means using :attr:`scanpy._settings.ScanpyConfig.n_jobs`.
    copy
        Determines whether a copy of `adata` is returned.

    Returns
    -------
    Depending on `copy` returns or updates `adata` with the corrected data matrix.
    zregressing out z=    sparse input is densified and may lead to high memory useNFr   r   zwIf providing categorical variable, only a single one is allowed. For this one we regress on the mean for each category.z2... regressing on per-gene means within categoriesrX   )rW   TÚonesg      ð?iè  r2   )ÚParallelÚdelayed)r|   c                 3   s   | ]}ˆ t ƒ|ƒV  qd S r*   )Ú_regress_out_chunk)r-   Útask©r   r+   r/   r0   š  s     zregress_out.<locals>.<genexpr>z    finishedrr   ))r5   r@   r
   r;   r'   r   Zis_viewZ_init_as_actualr9   ÚstrÚtoarrayÚsettr|   Zobs_keysr   r<   Úlenr8   Údebugr>   ÚzerosÚshapeÚcatÚ
categoriesÚvaluesÚ	enumerateÚTÚmeanÚinsertÚceilÚminr]   ÚintZarray_splitÚappendÚtupleZjoblibr~   r   ÚvstackrW   )rB   rv   r|   r'   rf   Úvariable_is_categoricalÚ
regressorsÚcategoryÚmaskZixÚxZ	len_chunkZn_chunksÚtasksZ
chunk_listZregressors_chunkÚidxÚ
data_chunkÚregresr~   Úresr+   r‚   r/   Úregress_out:  sZ    



&ÿ

"
r¡   c              	   C   s&  | d }| d }| d }g }dd l m} ddlm} t|jd ƒD ]Ø}|d d …|f |d|f k ¡ s~| |d d …|f ¡ qB|rªtj	t 
|jd ¡|d d …|f f }n|}z0|j|d d …|f ||j ¡ d� ¡ }	|	j}
W n0 |k
�r   t d¡ t |jd ¡}
Y nX | |
¡ qBt |¡S )Nr   r   r   )ÚPerfectSeparationError)Úfamilyz9Encountered PerfectSeparationError, setting to 0 as in R.)Zstatsmodels.apiÚapiZstatsmodels.tools.sm_exceptionsr¢   Úranger‰   Úanyr”   r>   Zc_r}   ZGLMZfamiliesZGaussianÚfitZresid_responser5   r6   rˆ   r–   )r!   rž   r˜   r—   Zresponses_chunk_listÚsmr¢   Z	col_indexrŸ   ÚresultZ
new_columnr+   r+   r/   r€   £  s2     (  ÿ


r€   ©r;   Úzero_centerÚ	max_valuer'   rP   rQ   c                 C   sP   t ||d� |dk	r&tdt| ƒ› �ƒ‚|dk	r@tdt| ƒ› �ƒ‚t| |||d�S )uQ      Scale data to unit variance and zero mean.

    .. note::
        Variables (genes) that do not display any variation (are constant across
        all observations) are retained and (for zero_center==True) set to 0
        during this operation. In the future, they might be set to NaNs.

    Parameters
    ----------
    X
        The (annotated) data matrix of shape `n_obs` Ã— `n_vars`.
        Rows correspond to cells and columns to genes.
    zero_center
        If `False`, omit zero-centering variables, which allows to handle sparse
        input efficiently.
    max_value
        Clip (truncate) to this value after scaling. If `None`, do not clip.
    copy
        Whether this function should be performed inplace. If an AnnData object
        is passed, this also determines if a copy is returned.
    layer
        If provided, which element of layers to scale.
    obsm
        If provided, which element of obsm to scale.

    Returns
    -------
    Depending on `copy` returns or updates `adata` with a scaled `adata.X`,
    annotated with `'mean'` and `'std'` in `adata.var`.
    rb   Nz1`layer` argument inappropriate for value of type z0`obsm` argument inappropriate for value of type )r«   r¬   r'   )r   r8   ÚtypeÚscale_arrayrª   r+   r+   r/   ÚscaleÈ  s    (r¯   ©r«   r¬   r'   Úreturn_mean_stdc                C   sÜ   |r|   ¡ } |s"|d k	r"t d¡ t | jtj¡rFt d¡ |  t¡} t	| ƒ\}}t 
|¡}d||dk< t| ƒrŽ|r|tdƒ‚t | d| ¡ n|rš| |8 } | | } |d k	rÆt d|› �¡ || | |k< |rÔ| ||fS | S d S )Nz<... be careful when using `max_value` without `zero_center`.zV... as scaling leads to float results, integer input is cast to float, returning copy.r   r   z!Cannot zero-center sparse matrix.z... clipping at max_value )r'   r5   r@   r>   r\   rW   Úintegerr]   r^   r   ri   r
   r8   r   Zinplace_column_scaler‡   )r;   r«   r¬   r'   r±   r�   rL   Ústdr+   r+   r/   r®   ø  s6    	ÿÿ


r®   c                C   s,   |rt  d¡ |  ¡ } d}t| ||||d�S )Nz]... as `zero_center=True`, sparse input is densified and may lead to large memory consumptionF)r«   r'   r¬   r±   )r5   r@   r„   r®   )r;   r«   r¬   r'   r±   r+   r+   r/   Úscale_sparse&  s    
ÿûr´   )r«   r¬   r'   rP   rQ   )rB   r«   r¬   r'   rP   rQ   r(   c                C   sf   |r|   ¡ n| } t| ƒ t| ||d�}t|||ddd�\}| jd< | jd< t| |||d� |rb| S d S )Nrb   FTr°   r�   r³   )r'   r   r   r¯   rL   r   )rB   r«   r¬   r'   rP   rQ   r;   r+   r+   r/   Úscale_anndata@  s    
ûrµ   )r!   ÚfractionÚn_obsÚrandom_stater'   r(   c           	      C   sÎ   t j |¡ t| tƒr| jn| jd }|dk	r4|}nN|dk	rz|dksL|dk rZtd|› �ƒ‚t|| ƒ}t	 
d|› d�¡ ntdƒ‚t jj||dd	�}t| tƒrº|r®| |  ¡ S |  |¡ n| }|| |fS dS )
u÷      Subsample to a fraction of the number of observations.

    Parameters
    ----------
    data
        The (annotated) data matrix of shape `n_obs` Ã— `n_vars`.
        Rows correspond to cells and columns to genes.
    fraction
        Subsample to this `fraction` of the number of observations.
    n_obs
        Subsample to this number of observations.
    random_state
        Random seed to change subsampling.
    copy
        If an :class:`~anndata.AnnData` is passed,
        determines whether a copy is returned.

    Returns
    -------
    Returns `X[obs_indices], obs_indices` if data is array-like, otherwise
    subsamples the passed :class:`~anndata.AnnData` (`copy == False`) or
    returns a subsampled copy of it (`copy == True`).
    r   Nr   z*`fraction` needs to be within [0, 1], not z... subsampled to z data pointsz"Either pass `n_obs` or `fraction`.F)ÚsizeÚreplace)r>   ÚrandomÚseedr9   r   r·   r‰   r8   r“   r5   r‡   Úchoicer'   r=   )	r!   r¶   r·   r¸   r'   Z	old_n_obsZ	new_n_obsZobs_indicesr;   r+   r+   r/   Ú	subsampleY  s"    
r¾   Ztarget_countsrm   )r¸   rº   r'   )rB   rm   Útotal_countsr¸   rº   r'   r(   c                C   sf   |dk	}|dk	}||kr t dƒ‚|r,|  ¡ } |rDt| j|||ƒ| _n|rZt| j|||ƒ| _|rb| S dS )a      Downsample counts from count matrix.

    If `counts_per_cell` is specified, each cell will downsampled.
    If `total_counts` is specified, expression matrix will be downsampled to
    contain at most `total_counts`.

    Parameters
    ----------
    adata
        Annotated data matrix.
    counts_per_cell
        Target total counts per cell. If a cell has more than 'counts_per_cell',
        it will be downsampled to this number. Resulting counts can be specified
        on a per cell basis by passing an array.Should be an integer or integer
        ndarray with same length as number of obs.
    total_counts
        Target total counts. If the count matrix has more than `total_counts`
        it will be downsampled to have this number.
    random_state
        Random seed for subsampling.
    replace
        Whether to sample the counts with replacement.
    copy
        Determines whether a copy of `adata` is returned.

    Returns
    -------
    Depending on `copy` returns or updates an `adata` with downsampled `.X`.
    Nz@Must specify exactly one of `total_counts` or `counts_per_cell`.)r8   r'   Ú_downsample_total_countsr;   Ú_downsample_per_cell)rB   rm   r¿   r¸   rº   r'   Ztotal_counts_callZcounts_per_cell_callr+   r+   r/   Údownsample_countsŽ  s    )ÿrÂ   c                 C   sT  | j d }t|tƒr"t ||¡}n
t |¡}|jtjdd�}t|tjƒrTt	|ƒ|kr\t
dƒ‚t| ƒrút| ƒ}t| ƒs|t| ƒ} t | jdd�¡}t ||k¡d }t | j| jdd… ¡}|D ]"}	||	 }
t|
||	 ||dd	� q¼|  ¡  |tk	rø|| ƒ} nVt | jdd�¡}t ||k¡d }|D ],}	| |	d d …f }
t|
||	 ||dd	� �q"| S )
Nr   Frq   zŸIf provided, 'counts_per_cell' must be either an integer, or coercible to an `np.ndarray` of length as number of observations by `np.asarray(counts_per_cell)`.r   r2   éÿÿÿÿT©r¸   rº   r&   )r‰   r9   r“   r>   ÚfullZasarrayr]   Úint_Úndarrayr†   r8   r
   r­   r   r   Zravelr7   ZnonzeroÚsplitr!   ZindptrÚ_downsample_arrayÚeliminate_zeros)r;   rm   r¸   rº   r·   Úoriginal_typeZtotalsZunder_targetÚrowsZrowidxÚrowr+   r+   r/   rÁ   Ç  sP    


ÿû
û
rÁ   c                 C   s’   t |ƒ}|  ¡ }||k r| S t| ƒrjt| ƒ}t| ƒs<t| ƒ} t| j|||dd� |  ¡  |tk	rŽ|| ƒ} n$|  	t
j| jŽ ¡}t||||dd� | S )NTrÄ   )rº   r&   )r“   r7   r
   r­   r   r   rÉ   r!   rÊ   Zreshaper>   Úmultiplyr‰   )r;   r¿   r¸   rº   ÚtotalrË   Úvr+   r+   r/   rÀ   ÷  s*    û
rÀ   )Úcache)ÚcolÚtargetr¸   rº   r&   c           
      C   s�   t j |¡ |  ¡ }|r&d| dd…< n
t  | ¡} t  |d ¡}t jj|||d�}| ¡  d}|D ]*}	|	|| krz|d7 }qd| |  d7  < q`| S )z©    Evenly reduce counts in cell to target amount.

    This is an internal function and has some restrictions:

    * total counts in cell must be less than target
    r   NrÃ   )rº   r   )r>   r»   r¼   ZcumsumZ
zeros_likerÆ   r½   Úsort)
rÒ   rÓ   r¸   rº   r&   Z	cumcountsrÏ   ÚsampleZgeneptrÚcountr+   r+   r/   rÉ     s    

rÉ   c                 C   s†   | | j dd�8 } tj| dd�}tjjj||d�\}}t |¡d d d… }|d d …|f }|| }|d d …d |…f }t |j	| j	¡j	S )Nr   r2   F)Zrowvar)ÚkrÃ   )
r�   r>   ZcovÚspÚsparseZlinalgZeigshZargsortÚdotrŽ   )r!   Zn_compsÚCZevalsZevecsZidcsr+   r+   r/   Ú_pca_fallback5  s    rÜ   )NNNNTF)NNNNTF)FFN)NNr1   Fr+   Nr   )NF)TNFNN)NNr   F)NN)r   TF)r   )SÚ__doc__Ú	functoolsr   Únumbersr   rw   Útypingr   r   r   r   r   r	   ZnumbaÚnumpyr>   ZscipyrØ   Zscipy.sparser
   r   r   r   Zsklearn.utilsr   r   Zpandas.api.typesr   Zanndatar   Ú r   r5   Z	_settingsr   r…   Ú_utilsr   r   r   r   r   Z_compatr   Úgetr   r   Z_distributedr   r   Z
dask.arrayÚarrayÚdaÚImportErrorZ!_deprecated.highly_variable_genesr    r“   ÚboolrÇ   r:   rK   rƒ   rU   ÚregisterrZ   rT   rh   ri   r^   ru   r¡   r€   r¯   r®   r´   rµ   r¾   rÂ   rÁ   rÀ   ZnjitrÉ   rÜ   r+   r+   r+   r/   Ú<module>   sÀ   
      ùø 
      ùøløø.
ø÷#   üû.       ø
÷ 
  üûi%     úú/
úú-úúùø    ûú5
  ýùø80
   ûû$