U
    ½mœdíG  ã                   @   sÄ   d Z ddlZddlZddlmZ ddlmZ ddlmZ ddl	Z
ddlZddlmZ ddlmZ dd	lmZmZmZ eeed
œdd„Zeee
jd
œdd„Zdd„ Zddd„Zddd„Zddd„ZdS )z9Implementation of ARFF parsers: via LIAC-ARFF and pandas.é    N)ÚOrderedDict)Ú	Generator)ÚListé   )Ú_arff)ÚArffSparseDataType)Ú_chunk_generatorÚcheck_pandas_supportÚget_chunk_n_rows)Ú	arff_dataÚinclude_columnsÚreturnc                 C   s€   t ƒ t ƒ t ƒ f}dd„ t|ƒD ƒ}t| d | d | d ƒD ]@\}}}||kr:|d  |¡ |d  |¡ |d  || ¡ q:|S )a™  Obtains several columns from sparse ARFF representation. Additionally,
    the column indices are re-labelled, given the columns that are not
    included. (e.g., when including [1, 2, 3], the columns will be relabelled
    to [0, 1, 2]).

    Parameters
    ----------
    arff_data : tuple
        A tuple of three lists of equal size; first list indicating the value,
        second the x coordinate and the third the y coordinate.

    include_columns : list
        A list of columns to include.

    Returns
    -------
    arff_data_new : tuple
        Subset of arff data with only the include columns indicated by the
        include_columns argument.
    c                 S   s   i | ]\}}||“qS © r   ©Ú.0Z	array_idxZ
column_idxr   r   úV/home/sam/Atlas/atlas_env/lib/python3.8/site-packages/sklearn/datasets/_arff_parser.pyÚ
<dictcomp>-   s     z)_split_sparse_columns.<locals>.<dictcomp>r   é   r   )ÚlistÚ	enumerateÚzipÚappend)r   r   Zarff_data_newÚreindexed_columnsÚvalÚrow_idxÚcol_idxr   r   r   Ú_split_sparse_columns   s    ÿ"r   c           	      C   s~   t | d ƒd }|t|ƒf}dd„ t|ƒD ƒ}tj|tjd�}t| d | d | d ƒD ]"\}}}||krV||||| f< qV|S )Nr   c                 S   s   i | ]\}}||“qS r   r   r   r   r   r   r   ?   s     z)_sparse_data_to_array.<locals>.<dictcomp>©Údtyper   r   )ÚmaxÚlenr   ÚnpÚemptyÚfloat64r   )	r   r   Únum_obsZy_shaper   Úyr   r   r   r   r   r   Ú_sparse_data_to_array8   s    ÿ"r&   c                 C   sD   | | }t |ƒdkr| | }nt |ƒdkr8| |d  }nd}||fS )a  Post process a dataframe to select the desired columns in `X` and `y`.

    Parameters
    ----------
    frame : dataframe
        The dataframe to split into `X` and `y`.

    feature_names : list of str
        The list of feature names to populate `X`.

    target_names : list of str
        The list of target names to populate `y`.

    Returns
    -------
    X : dataframe
        The dataframe containing the features.

    y : {series, dataframe} or None
        The series or dataframe containing the target.
    r   r   r   N)r    )ÚframeZfeature_namesZtarget_namesÚXr%   r   r   r   Ú_post_process_frameJ   s    
r)   c           "         sh  dd„ }|| ƒ}|dkrt jnt j}|dk }	t j|||	d�}
|| ‰‡fdd„|
d D ƒ‰ |dk�rŽtd	ƒ}t|
d ƒ}t| ¡ ƒ}t|
d
 ƒ}|j	|g|d�}|j
dd� ¡ }t|ƒ}‡fdd„|D ƒ}|| g}t|
d
 |ƒD ]}| |j	||d�| ¡ qä|j|dd�}~~i }|jD ]P}ˆ| d }| ¡ dk�rFd||< n&| ¡ dk�r^d||< n|j| ||< �q| |¡}t|||ƒ\}‰�n¸|
d
 }‡fdd„|D ƒ}‡fdd„|D ƒ}t|tƒ�r@|dk�rØtdƒ‚|d dk�rìd}n|d |d  }tjtj |¡d|d�}|j|Ž }|dd…|f }|dd…|f ‰n€t|tƒ�r®t||ƒ}t |d ƒd }|t!|ƒf} t"j#j$|d |d |d ff| tj%d �}| &¡ }t'||ƒ‰ntd!t(|ƒ› �ƒ‚‡ fd"d#„|D ƒ}!|!�sÚn<t)|!ƒ�rt *‡ ‡fd$d„t+|ƒD ƒ¡‰nt,|!ƒ�rtd%ƒ‚ˆj-d dk�r2ˆ d&¡‰nˆj-d dk�rFd‰|dk�r\|ˆ|dfS |ˆdˆ fS )'a  ARFF parser using the LIAC-ARFF library coded purely in Python.

    This parser is quite slow but consumes a generator. Currently it is needed
    to parse sparse datasets. For dense datasets, it is recommended to instead
    use the pandas-based parser, although it does not always handles the
    dtypes exactly the same.

    Parameters
    ----------
    gzip_file : GzipFile instance
        The file compressed to be read.

    output_arrays_type : {"numpy", "sparse", "pandas"}
        The type of the arrays that will be returned. The possibilities ara:

        - `"numpy"`: both `X` and `y` will be NumPy arrays;
        - `"sparse"`: `X` will be sparse matrix and `y` will be a NumPy array;
        - `"pandas"`: `X` will be a pandas DataFrame and `y` will be either a
          pandas Series or DataFrame.

    columns_info : dict
        The information provided by OpenML regarding the columns of the ARFF
        file.

    feature_names_to_select : list of str
        A list of the feature names to be selected.

    target_names_to_select : list of str
        A list of the target names to be selected.

    Returns
    -------
    X : {ndarray, sparse matrix, dataframe}
        The data matrix.

    y : {ndarray, dataframe, series}
        The target.

    frame : dataframe or None
        A dataframe containing both `X` and `y`. `None` if
        `output_array_type != "pandas"`.

    categories : list of str or None
        The names of the features that are categorical. `None` if
        `output_array_type == "pandas"`.
    c                 s   s   | D ]}|  d¡V  qd S )Núutf-8)Údecode)Ú	gzip_fileÚliner   r   r   Ú_io_to_generator¡   s    z+_liac_arff_parser.<locals>._io_to_generatorÚsparseÚpandas)Úreturn_typeÚencode_nominalc                    s(   i | ] \}}t |tƒr|ˆ kr||“qS r   )Ú
isinstancer   )r   ÚnameÚcat©Úcolumns_to_selectr   r   r   ±   s
   
 þ z%_liac_arff_parser.<locals>.<dictcomp>Ú
attributeszfetch_openml with as_frame=TrueÚdata)ÚcolumnsT)Údeepc                    s   g | ]}|ˆ kr|‘qS r   r   ©r   Úcolr6   r   r   Ú
<listcomp>Ä   s      z%_liac_arff_parser.<locals>.<listcomp>)Zignore_indexÚ	data_typeÚintegerÚInt64ÚnominalÚcategoryc                    s   g | ]}t ˆ | d  ƒ‘qS ©Úindex©Úint©r   Úcol_name©Úopenml_columns_infor   r   r>   ß   s   ÿc                    s   g | ]}t ˆ | d  ƒ‘qS rD   rF   rH   rJ   r   r   r>   ã   s   ÿNz6shape must be provided when arr['data'] is a Generatorr   éÿÿÿÿr   r#   )r   Úcountr   )Úshaper   z-Unexpected type for data obtained from arff: c                    s   h | ]}|ˆ k’qS r   r   rH   )Ú
categoriesr   r   Ú	<setcomp>
  s    z$_liac_arff_parser.<locals>.<setcomp>c              
      sJ   g | ]B\}}t  t jˆ  |¡d d�ˆdd…||d …f jtdd�¡‘qS )ÚOr   Nr   F)Úcopy)r!   ZtakeZasarrayÚpopÚastyperG   )r   ÚirI   )rO   r%   r   r   r>     s
   ü þzAMix of nominal and non-nominal targets is not currently supported)rL   ).r   ZCOOZ	DENSE_GENÚloadr	   r   r   ÚkeysÚnextZ	DataFrameZmemory_usageÚsumr
   r   r   Úconcatr:   ÚlowerÚdtypesrT   r)   r3   r   Ú
ValueErrorr!   ZfromiterÚ	itertoolsÚchainÚfrom_iterableZreshapeÚtupler   r   r    Úspr/   Z
coo_matrixr#   Ztocsrr&   ÚtypeÚallZhstackr   ÚanyrN   )"r,   Úoutput_arrays_typerK   Úfeature_names_to_selectÚtarget_names_to_selectrN   r.   Ústreamr1   r2   Zarff_containerÚpdZcolumns_infoZcolumns_namesÚ	first_rowZfirst_dfZ	row_bytesÚ	chunksizeÚcolumns_to_keepÚdfsr9   r'   r\   r4   Úcolumn_dtyper(   r   Zfeature_indices_to_selectZtarget_indices_to_selectrM   Zarff_data_Xr$   ZX_shapeZis_classificationr   )rO   r7   rK   r%   r   Ú_liac_arff_parserj   sÈ    7
  ÿ
þ





  ÿ
þ
þ
ÿ
ý

ýÿ
ÿ
ûÿ	
ÿ
rp   c              
      sÊ  ddl ‰| D ]}| d¡ ¡  d¡r q*qi ‰|D ]:}|| d }| ¡ dkrXdˆ|< q2| ¡ dkr2d	ˆ|< q2‡fd
d„t|ƒD ƒ}	dddgdddd|	dœ}
|
|p¤i –}ˆj| f|Ž}zdd„ |D ƒ|_W n0 tk
rú } zˆj 	d¡|‚W 5 d}~X Y nX || ‰ ‡ fdd„|jD ƒ}|| }t
 d¡‰‡fdd„}‡fdd„|j ¡ D ƒ}|D ]}|| j |¡||< �qRt|||ƒ\}}|dk�r”|||dfS | ¡ | ¡  }}‡fdd„|j ¡ D ƒ}||d|fS )a^  ARFF parser using `pandas.read_csv`.

    This parser uses the metadata fetched directly from OpenML and skips the metadata
    headers of ARFF file itself. The data is loaded as a CSV file.

    Parameters
    ----------
    gzip_file : GzipFile instance
        The GZip compressed file with the ARFF formatted payload.

    output_arrays_type : {"numpy", "sparse", "pandas"}
        The type of the arrays that will be returned. The possibilities are:

        - `"numpy"`: both `X` and `y` will be NumPy arrays;
        - `"sparse"`: `X` will be sparse matrix and `y` will be a NumPy array;
        - `"pandas"`: `X` will be a pandas DataFrame and `y` will be either a
          pandas Series or DataFrame.

    openml_columns_info : dict
        The information provided by OpenML regarding the columns of the ARFF
        file.

    feature_names_to_select : list of str
        A list of the feature names to be selected to build `X`.

    target_names_to_select : list of str
        A list of the target names to be selected to build `y`.

    read_csv_kwargs : dict, default=None
        Keyword arguments to pass to `pandas.read_csv`. It allows to overwrite
        the default options.

    Returns
    -------
    X : {ndarray, sparse matrix, dataframe}
        The data matrix.

    y : {ndarray, dataframe, series}
        The target.

    frame : dataframe or None
        A dataframe containing both `X` and `y`. `None` if
        `output_array_type != "pandas"`.

    categories : list of str or None
        The names of the features that are categorical. `None` if
        `output_array_type == "pandas"`.
    r   Nr*   z@datar?   r@   rA   rB   rC   c                    s"   i | ]\}}|ˆ kr|ˆ | “qS r   r   )r   r   r4   )r\   r   r   r   u  s   þ z'_pandas_arff_parser.<locals>.<dictcomp>Fú?ú%ú"Tú\)ÚheaderZ	index_colZ	na_valuesÚcommentÚ	quotecharÚskipinitialspaceÚ
escapecharr   c                 S   s   g | ]}|‘qS r   r   )r   r4   r   r   r   r>   Œ  s     z'_pandas_arff_parser.<locals>.<listcomp>zwThe number of columns provided by OpenML does not match the number of columns inferred by pandas when reading the file.c                    s   g | ]}|ˆ kr|‘qS r   r   r<   r6   r   r   r>   ”  s      z^'(?P<contents>.*)'$c                    s"   t  ˆ | ¡}|d kr| S | d¡S )NÚcontents)ÚreÚsearchÚgroup)Zinput_stringÚmatch)Úsingle_quote_patternr   r   Ústrip_single_quotes¤  s    z0_pandas_arff_parser.<locals>.strip_single_quotesc                    s"   g | ]\}}ˆ j j |¡r|‘qS r   )ÚapiÚtypesÚis_categorical_dtype©r   r4   r   ©rj   r   r   r>   «  s   þr0   c                    s*   i | ]"\}}ˆ j j |¡r||j ¡ “qS r   )r�   r‚   rƒ   rO   Útolistr„   r…   r   r   r   º  s   þ )r0   r+   r[   Ú
startswithr   Zread_csvr:   r]   ÚerrorsZParserErrorr{   Úcompiler\   Úitemsr5   Zrename_categoriesr)   Zto_numpy)r,   rf   rK   rg   rh   Úread_csv_kwargsr-   r4   ro   Zdtypes_positionalZdefault_read_csv_kwargsr'   Úexcrm   r€   Zcategorical_columnsr=   r(   r%   rO   r   )r7   r\   rj   r   r   Ú_pandas_arff_parser+  sf    8


þø
ÿý

þ

þr�   c                 C   sH   |dkrt | |||||ƒS |dkr4t| |||||ƒS td|› d�ƒ‚dS )a6  Load a compressed ARFF file using a given parser.

    Parameters
    ----------
    gzip_file : GzipFile instance
        The file compressed to be read.

    parser : {"pandas", "liac-arff"}
        The parser used to parse the ARFF file. "pandas" is recommended
        but only supports loading dense datasets.

    output_type : {"numpy", "sparse", "pandas"}
        The type of the arrays that will be returned. The possibilities ara:

        - `"numpy"`: both `X` and `y` will be NumPy arrays;
        - `"sparse"`: `X` will be sparse matrix and `y` will be a NumPy array;
        - `"pandas"`: `X` will be a pandas DataFrame and `y` will be either a
          pandas Series or DataFrame.

    openml_columns_info : dict
        The information provided by OpenML regarding the columns of the ARFF
        file.

    feature_names_to_select : list of str
        A list of the feature names to be selected.

    target_names_to_select : list of str
        A list of the target names to be selected.

    read_csv_kwargs : dict, default=None
        Keyword arguments to pass to `pandas.read_csv`. It allows to overwrite
        the default options.

    Returns
    -------
    X : {ndarray, sparse matrix, dataframe}
        The data matrix.

    y : {ndarray, dataframe, series}
        The target.

    frame : dataframe or None
        A dataframe containing both `X` and `y`. `None` if
        `output_array_type != "pandas"`.

    categories : list of str or None
        The names of the features that are categorical. `None` if
        `output_array_type == "pandas"`.
    z	liac-arffr0   zUnknown parser: 'z%'. Should be 'liac-arff' or 'pandas'.N)rp   r�   r]   )r,   ÚparserÚoutput_typerK   rg   rh   rN   r‹   r   r   r   Úload_arff_from_gzip_fileÂ  s*    ;úú	
ÿr�   )N)N)NN)Ú__doc__r^   r{   Úcollectionsr   Úcollections.abcr   Útypingr   Únumpyr!   Zscipyrb   Z	externalsr   Zexternals._arffr   Úutilsr   r	   r
   r   Zndarrayr&   r)   rp   r�   r�   r   r   r   r   Ú<module>   s8    þ$ þ& ú
 H ú
   ø