U
    ÅmœdÂ  ã                   @   sÂ   d Z ddlmZmZmZmZ ddlZddlm	Z	 ddl
mZ ddlmZ ddlmZ dd	lmZmZmZ eeed
�de	eee ee eee  ee eeeee eeeeef  dœdd„ƒZdS )zA
Computes a dendrogram based on a given categorical observation.
é    )ÚOptionalÚSequenceÚDictÚAnyN)ÚAnnData)Úis_categorical_dtypeé   )Úlogging)Ú_doc_params)Ú_choose_representationÚdoc_use_repÚ	doc_n_pcs)Ún_pcsÚuse_repÚpearsonÚcompleteFT)ÚadataÚgroupbyr   r   Ú	var_namesÚuse_rawÚ
cor_methodÚlinkage_methodÚoptimal_orderingÚ	key_addedÚinplaceÚreturnc                 C   s  t |tƒr|g}|D ]R}||  ¡ kr<td|› d|  ¡ › �ƒ‚t| j| ƒstd|› d| j| j› �ƒ‚q|dkrøt t	| ||d�¡}| j|d  }t
|ƒdkrÔ|dd… D ](}| t¡d	 | j|  t¡  d
¡}qªd	 |¡|_|j|dd� |jj}n2|�r| jjn| j}ddlm} || |||ƒ\}}|jdd� ¡ }ddlm  m} ddlm} |jj|d�}| d| ¡}|j|||d�}|j |t!|ƒdd�}t"||||||d |d ||j#d�	}|
�rú|	dk�rÜdd	 |¡› �}	t$ %d|	›d�¡ || j&|	< n|S dS )a§	      Computes a hierarchical clustering for the given `groupby` categories.

    By default, the PCA representation is used unless `.X`
    has less than 50 variables.

    Alternatively, a list of `var_names` (e.g. genes) can be given.

    Average values of either `var_names` or components are used
    to compute a correlation matrix.

    The hierarchical clustering can be visualized using
    :func:`scanpy.pl.dendrogram` or multiple other visualizations that can
    include a dendrogram: :func:`~scanpy.pl.matrixplot`,
    :func:`~scanpy.pl.heatmap`, :func:`~scanpy.pl.dotplot`,
    and :func:`~scanpy.pl.stacked_violin`.

    .. note::
        The computation of the hierarchical clustering is based on predefined
        groups and not per cell. The correlation matrix is computed using by
        default pearson but other methods are available.

    Parameters
    ----------
    adata
        Annotated data matrix
    {n_pcs}
    {use_rep}
    var_names
        List of var_names to use for computing the hierarchical clustering.
        If `var_names` is given, then `use_rep` and `n_pcs` is ignored.
    use_raw
        Only when `var_names` is not None.
        Use `raw` attribute of `adata` if present.
    cor_method
        correlation method to use.
        Options are 'pearson', 'kendall', and 'spearman'
    linkage_method
        linkage method to use. See :func:`scipy.cluster.hierarchy.linkage`
        for more information.
    optimal_ordering
        Same as the optimal_ordering argument of :func:`scipy.cluster.hierarchy.linkage`
        which reorders the linkage matrix so that the distance between successive
        leaves is minimal.
    key_added
        By default, the dendrogram information is added to
        `.uns[f'dendrogram_{{groupby}}']`.
        Notice that the `groupby` information is added to the dendrogram.
    inplace
        If `True`, adds dendrogram information to `adata.uns[key_added]`,
        else this function returns the information.

    Returns
    -------
    If `inplace=False`, returns dendrogram information,
    else `adata.uns[key_added]` is updated with it.

    Examples
    --------
    >>> import scanpy as sc
    >>> adata = sc.datasets.pbmc68k_reduced()
    >>> sc.tl.dendrogram(adata, groupby='bulk_labels')
    >>> sc.pl.dendrogram(adata)
    >>> markers = ['C1QA', 'PSAP', 'CD79A', 'CD79B', 'CST3', 'LYZ']
    >>> sc.pl.dotplot(adata, markers, groupby='bulk_labels', dendrogram=True)
    z4groupby has to be a valid observation. Given value: z, valid observations: z:groupby has to be a categorical observation. Given value: z, Column type: N)r   r   r   é   Ú_ÚcategoryT)r   r   )Ú_prepare_dataframe)Úlevel)Údistance)Úmethod)r"   r   )ÚlabelsZno_plotZivlÚleaves)	Úlinkager   r   r   r   Zcategories_orderedZcategories_idx_orderedZdendrogram_infoZcorrelation_matrixZdendrogram_z$Storing dendrogram info using `.uns[z]`)'Ú
isinstanceÚstrZobs_keysÚ
ValueErrorr   ZobsZdtypeÚpdZ	DataFramer   ÚlenZastypeÚjoinÚnameZ	set_indexÚindexÚ
categoriesÚrawr   Zplotting._anndatar   r   ZmeanZscipy.cluster.hierarchyZclusterZ	hierarchyZscipy.spatialr!   ÚTZcorrZ
squareformr%   Ú
dendrogramÚlistÚdictÚvaluesÚloggÚinfoZuns)r   r   r   r   r   r   r   r   r   r   r   ÚgroupZrep_dfZcategoricalr.   Z
gene_namesr   Zmean_dfZschr!   Zcorr_matrixZcorr_condensedZz_varZdendro_infoZdat© r8   úQ/home/sam/Atlas/atlas_env/lib/python3.8/site-packages/scanpy/tools/_dendrogram.pyr1      sp    P
ÿÿÿÿþ
  ÿ÷
r1   )	NNNNr   r   FNT)Ú__doc__Útypingr   r   r   r   Zpandasr)   Zanndatar   Zpandas.api.typesr   Ú r	   r5   Ú_utilsr
   Ztools._utilsr   r   r   r'   ÚintÚboolr1   r8   r8   r8   r9   Ú<module>   s>   
         õ
ô