Ë
    ÷Q(hØJ  ã                   óî   — d Z ddlZddlZddlmZ ddlmZ ddlmZ ddl	Z
ddlZddlmZ ddlmZ dd	lmZmZ dd
lmZ ddlmZ dededefd„Zdedede
j2                  fd„Zd„ Z	 dd„Z	 dd„Z	 	 dd„Zy)z9Implementation of ARFF parsers: via LIAC-ARFF and pandas.é    N)ÚOrderedDict)Ú	Generator)ÚListé   )Ú_arff)ÚArffSparseDataType)Úchunk_generatorÚget_chunk_n_rows)Úcheck_pandas_support)Ú	pd_fillnaÚ	arff_dataÚinclude_columnsÚreturnc                 óN  — t        «       t        «       t        «       f}t        |«      D ��ci c]  \  }}||“Œ
 }}}t        | d   | d   | d   «      D ]J  \  }}}||v sŒ|d   j                  |«       |d   j                  |«       |d   j                  ||   «       ŒL |S c c}}w )a™  Obtains several columns from sparse ARFF representation. Additionally,
    the column indices are re-labelled, given the columns that are not
    included. (e.g., when including [1, 2, 3], the columns will be relabelled
    to [0, 1, 2]).

    Parameters
    ----------
    arff_data : tuple
        A tuple of three lists of equal size; first list indicating the value,
        second the x coordinate and the third the y coordinate.

    include_columns : list
        A list of columns to include.

    Returns
    -------
    arff_data_new : tuple
        Subset of arff data with only the include columns indicated by the
        include_columns argument.
    r   é   r   )ÚlistÚ	enumerateÚzipÚappend)	r   r   Úarff_data_newÚ	array_idxÚ
column_idxÚreindexed_columnsÚvalÚrow_idxÚcol_idxs	            ú[/var/www/skyplay_api_hub/venv/lib/python3.12/site-packages/sklearn/datasets/_arff_parser.pyÚ_split_sparse_columnsr      sÂ   € ô. *.«´³¼»Ð(@€Mä;DÀ_Ó;U÷Ù"7 )¨Zˆ
�IÑðÐñ ô "% Y¨q¡\°9¸Q±<ÀÈ1ÁÓ!Nò @ÑˆˆW�gØ�oÒ%Ø˜!Ñ×#Ñ# CÔ(Ø˜!Ñ×#Ñ# GÔ,Ø˜!Ñ×#Ñ#Ð$5°gÑ$>Õ?ð	@ð
 Ðùós   ¬B!c                 ó0  — t        | d   «      dz   }|t        |«      f}t        |«      D ��ci c]  \  }}||“Œ
 }}}t        j                  |t        j
                  ¬«      }t        | d   | d   | d   «      D ]  \  }}	}
|
|v sŒ|||	||
   f<   Œ |S c c}}w )Nr   ©Údtyper   r   )ÚmaxÚlenr   ÚnpÚemptyÚfloat64r   )r   r   Únum_obsÚy_shaper   r   r   Úyr   r   r   s              r   Ú_sparse_data_to_arrayr*   9   s¹   € ô
 �)˜A‘,Ó !Ñ#€GØœ˜OÓ,Ð-€Gä;DÀ_Ó;U÷Ù"7 )¨Zˆ
�IÑðÐñ ô 	�‰�¤§
¡
Ô+€AÜ!$ Y¨q¡\°9¸Q±<ÀÈ1ÁÓ!Nò 9ÑˆˆW�gØ�oÒ%Ø58ˆAˆgÐ(¨Ñ1Ð1Ò2ð9ð €Hùós   ­Bc                 óz   — | |   }t        |«      dk\  r	| |   }||fS t        |«      dk(  r| |d      }||fS d}||fS )a  Post process a dataframe to select the desired columns in `X` and `y`.

    Parameters
    ----------
    frame : dataframe
        The dataframe to split into `X` and `y`.

    feature_names : list of str
        The list of feature names to populate `X`.

    target_names : list of str
        The list of target names to populate `y`.

    Returns
    -------
    X : dataframe
        The dataframe containing the features.

    y : {series, dataframe} or None
        The series or dataframe containing the target.
    r   r   r   N)r#   )ÚframeÚfeature_namesÚtarget_namesÚXr)   s        r   Ú_post_process_framer0   K   sh   € ð, 	ˆmÑ€AÜ
ˆ<Ó˜AÒØ�,Ñˆð
 ˆaˆ4€Kô	 
ˆ\Ó	˜aÒ	Ø�,˜q‘/Ñ"ˆð ˆaˆ4€Kð ˆØˆaˆ4€Kó    c                 óœ	  — d„ } || «      }|dk(  rt         j                  nt         j                  }|dk(   }	t        j                  |||	¬«      }
||z   }|
d   D ��ci c]  \  }}t	        |t
        «      r||v r||“Œ }}}|dk(  �r©t        d«      }t        |
d   «      }t        |j                  «       «      }t        |
d   «      }|j                  |g|d¬	«      }|j                  d
¬«      j                  «       }t        |«      }|D �cg c]	  }||v sŒ|‘Œ }}||   g}t        |
d   |«      D ](  }|j                  |j                  ||d¬	«      |   «       Œ* t!        |«      dk\  r$|d   j#                  |d   j$                  «      |d<   |j'                  |d
¬«      }t)        ||«      }~~i }|j*                  D ]N  }||   d   }|j-                  «       dk(  rd||<   Œ$|j-                  «       dk(  rd||<   Œ=|j$                  |   ||<   ŒP |j#                  |«      }t/        |||«      \  }}�nn|
d   }|D � cg c]  } t1        ||    d   «      ‘Œ }!} |D � cg c]  } t1        ||    d   «      ‘Œ }"} t	        |t2        «      rz|€t5        d«      ‚|d   dk(  rd}#n|d   |d   z  }#t7        j8                  t:        j<                  j?                  |«      d|#¬«      } |j@                  |Ž }|dd…|!f   }|dd…|"f   }n«t	        |tB        «      r„tE        ||!«      }$tG        |d   «      dz   }%|%t!        |!«      f}&tH        jJ                  jM                  |$d   |$d   |$d   ff|&t6        jN                  ¬«      }|jQ                  «       }tS        ||"«      }nt5        dtU        |«      › �«      ‚|D � ch c]  } | |v ’Œ }'} |'sn¬tW        |'«      r‹t7        jX                  t[        |«      D �(� cg c]`  \  }(} t7        j\                  t7        j^                  |ja                  | «      d¬«      |dd…|(|(dz   …f   j#                  t0        d¬«      «      ‘Œb c} }(«      }ntc        |'«      rt5        d «      ‚|jd                  d   dk(  r|jA                  d!«      }n|jd                  d   dk(  rd}|dk(  r||dfS ||d|fS c c}}w c c}w c c} w c c} w c c} w c c} }(w )"a  ARFF parser using the LIAC-ARFF library coded purely in Python.

    This parser is quite slow but consumes a generator. Currently it is needed
    to parse sparse datasets. For dense datasets, it is recommended to instead
    use the pandas-based parser, although it does not always handles the
    dtypes exactly the same.

    Parameters
    ----------
    gzip_file : GzipFile instance
        The file compressed to be read.

    output_arrays_type : {"numpy", "sparse", "pandas"}
        The type of the arrays that will be returned. The possibilities ara:

        - `"numpy"`: both `X` and `y` will be NumPy arrays;
        - `"sparse"`: `X` will be sparse matrix and `y` will be a NumPy array;
        - `"pandas"`: `X` will be a pandas DataFrame and `y` will be either a
          pandas Series or DataFrame.

    columns_info : dict
        The information provided by OpenML regarding the columns of the ARFF
        file.

    feature_names_to_select : list of str
        A list of the feature names to be selected.

    target_names_to_select : list of str
        A list of the target names to be selected.

    Returns
    -------
    X : {ndarray, sparse matrix, dataframe}
        The data matrix.

    y : {ndarray, dataframe, series}
        The target.

    frame : dataframe or None
        A dataframe containing both `X` and `y`. `None` if
        `output_array_type != "pandas"`.

    categories : list of str or None
        The names of the features that are categorical. `None` if
        `output_array_type == "pandas"`.
    c              3   ó@   K  — | D ]  }|j                  d«      –— Œ y ­w)Núutf-8)Údecode)Ú	gzip_fileÚlines     r   Ú_io_to_generatorz+_liac_arff_parser.<locals>._io_to_generator¢   s$   è ø€ Øò 	'ˆDØ—+‘+˜gÓ&Ó&ñ	'ùs   ‚ÚsparseÚpandas)Úreturn_typeÚencode_nominalÚ
attributeszfetch_openml with as_frame=TrueÚdataF)ÚcolumnsÚcopyT)Údeepr   r   r   )Úignore_indexÚ	data_typeÚintegerÚInt64ÚnominalÚcategoryÚindexNz6shape must be provided when arr['data'] is a Generatoréÿÿÿÿr&   )r!   Úcount)Úshaper!   z-Unexpected type for data obtained from arff: ÚOr    )r@   zAMix of nominal and non-nominal targets is not currently supported)rI   )3r   ÚCOOÚ	DENSE_GENÚloadÚ
isinstancer   r   r   ÚkeysÚnextÚ	DataFrameÚmemory_usageÚsumr
   r	   r   r#   ÚastypeÚdtypesÚconcatr   r?   Úlowerr0   Úintr   Ú
ValueErrorr$   ÚfromiterÚ	itertoolsÚchainÚfrom_iterableÚreshapeÚtupler   r"   Úspr9   Ú
coo_matrixr&   Útocsrr*   ÚtypeÚallÚhstackr   ÚtakeÚasarrayÚpopÚanyrK   ))r6   Úoutput_arrays_typeÚopenml_columns_infoÚfeature_names_to_selectÚtarget_names_to_selectrK   r8   Ústreamr;   r<   Úarff_containerÚcolumns_to_selectÚnameÚcatÚ
categoriesÚpdÚcolumns_infoÚcolumns_namesÚ	first_rowÚfirst_dfÚ	row_bytesÚ	chunksizeÚcolÚcolumns_to_keepÚdfsr>   r,   rW   Úcolumn_dtyper/   r)   r   Úcol_nameÚfeature_indices_to_selectÚtarget_indices_to_selectrJ   Úarff_data_Xr'   ÚX_shapeÚis_classificationÚis)                                            r   Ú_liac_arff_parserrˆ   k   sf  € òn'ñ ˜iÓ(€Fð  2°XÒ=”%—)’)Ä5Ç?Á?€Kð -°Ñ8Ð9€NÜ—Z‘ZØ˜K¸ô€Nð 0Ð2HÑHÐð (¨Ñ5÷áˆD�#Ü�cœ4Ô  TÐ->Ñ%>ð 	ˆc‰	ð€Jñ ð
 ˜XÓ%Ü!Ð"CÓDˆä" >°,Ñ#?Ó@ˆÜ˜\×.Ñ.Ó0Ó1ˆô ˜¨Ñ/Ó0ˆ	Ø—<‘<  °]È�<ÓOˆà×)Ñ)¨tÐ)Ó4×8Ñ8Ó:ˆ	Ü$ YÓ/ˆ	ð +8ÖT 3¸3ÐBSÒ;Sš3ÐTˆÐTØ˜Ñ(Ð)ˆÜ# N°6Ñ$:¸IÓFò 	ˆDØ�J‰JØ—‘˜T¨=¸u�ÓEÀoÑVõð	ô ˆs‹8�qŠ=Ø˜‘V—]‘] 3 q¡6§=¡=Ó1ˆC�‰Fð
 —	‘	˜#¨D�	Ó1ˆÜ˜"˜eÓ$ˆØ�ð ˆØ—M‘Mò 		2ˆDØ.¨tÑ4°[ÑAˆLØ×!Ñ!Ó# yÒ0ð  '��t’Ø×#Ñ#Ó%¨Ò2Ø)��t’à$Ÿ|™|¨DÑ1��t’ð		2ð —‘˜VÓ$ˆä"ØÐ*Ð,Bó
‰ˆŠ1ð # 6Ñ*ˆ	ð 4ö%
àô Ð# HÑ-¨gÑ6Õ7ð%
Ð!ð %
ð 3ö$
àô Ð# HÑ-¨gÑ6Õ7ð$
Ð ð $
ô
 �i¤Ô+Øˆ}Ü ØLóð ð �Q‰x˜2Š~Ø‘à˜a™ 5¨¡8Ñ+�Ü—;‘;Ü—‘×-Ñ-¨iÓ8ØØôˆDð
  �4—<‘< Ð'ˆDØ’QÐ1Ð1Ñ2ˆAØ’QÐ0Ð0Ñ1‰AÜ˜	¤5Ô)Ü/°	Ð;TÓUˆKÜ˜) A™,Ó'¨!Ñ+ˆGØ¤Ð$=Ó >Ð?ˆGÜ—	‘	×$Ñ$Ø˜Q‘ +¨a¡.°+¸a±.Ð!AÐBØÜ—j‘jð %ó ˆAð
 —‘“	ˆAÜ% iÐ1IÓJ‰Aô Ø?ÄÀYÃÐ?PÐQóð ð
 4Jö
Ø'/ˆH˜
Ò"ð
Ðð 
ñ !àÜÐ"Ô#Ü—	‘	ô (1Ð1GÓ'H÷ñ
 $˜˜8ô	 —G‘GÜŸ
™
 :§>¡>°(Ó#;À3ÔGØš!˜Q  Q¡˜Y˜,™×.Ñ.¬s¸Ð.Ó?õóó‰Aô Ð"Ô#ÜØSóð ð �7‰7�1‰:˜Š?Ø—	‘	˜%Ó ‰AØ�W‰W�Q‰Z˜1Š_ØˆAà˜XÒ%Ø�!�U˜DÐ Ð Øˆa��zÐ!Ð!ùóEùò& UùòL%
ùò$
ùòN
ùós+   Á!R.Ä
	R4ÄR4È8R9ÉR>Î)SÏ!A%S
c           
      óì  ‡— ddl }| D ]2  }|j                  d«      j                  «       j                  d«      sŒ2 n i }|D ]<  }	||	   d   }
|
j                  «       dk(  rd||	<   Œ$|
j                  «       dk(  sŒ8d	||	<   Œ> t	        |«      D ��	ci c]  \  }}	|	|v r|||	   “Œ }}}	dd
dgd
dddd|dœ	}i |¥|xs i ¥} |j
                  | fi |¤Ž}	 |D �	cg c]  }	|	‘Œ c}	|_        ||z   }|j                  D �cg c]	  }||v sŒ|‘Œ }}||   }t        j                  d«      Šˆfd„}|j                  j                  «       D �	�cg c]  \  }	}t        ||j                  «      r|	‘Œ }}	}|D ]#  }||   j                   j#                  |«      ||<   Œ% t%        |||«      \  }}|dk(  r|||dfS |j'                  «       |j'                  «       }}|j                  j                  «       D �	�ci c]6  \  }	}t        ||j                  «      r|	|j(                  j+                  «       “Œ8 }}	}||d|fS c c}	}w c c}	w # t        $ r!}|j                  j                  d«      |‚d}~ww xY wc c}w c c}}	w c c}}	w )a^  ARFF parser using `pandas.read_csv`.

    This parser uses the metadata fetched directly from OpenML and skips the metadata
    headers of ARFF file itself. The data is loaded as a CSV file.

    Parameters
    ----------
    gzip_file : GzipFile instance
        The GZip compressed file with the ARFF formatted payload.

    output_arrays_type : {"numpy", "sparse", "pandas"}
        The type of the arrays that will be returned. The possibilities are:

        - `"numpy"`: both `X` and `y` will be NumPy arrays;
        - `"sparse"`: `X` will be sparse matrix and `y` will be a NumPy array;
        - `"pandas"`: `X` will be a pandas DataFrame and `y` will be either a
          pandas Series or DataFrame.

    openml_columns_info : dict
        The information provided by OpenML regarding the columns of the ARFF
        file.

    feature_names_to_select : list of str
        A list of the feature names to be selected to build `X`.

    target_names_to_select : list of str
        A list of the target names to be selected to build `y`.

    read_csv_kwargs : dict, default=None
        Keyword arguments to pass to `pandas.read_csv`. It allows to overwrite
        the default options.

    Returns
    -------
    X : {ndarray, sparse matrix, dataframe}
        The data matrix.

    y : {ndarray, dataframe, series}
        The target.

    frame : dataframe or None
        A dataframe containing both `X` and `y`. `None` if
        `output_array_type != "pandas"`.

    categories : list of str or None
        The names of the features that are categorical. `None` if
        `output_array_type == "pandas"`.
    r   Nr4   z@datarC   rD   rE   rF   rG   Fú?ú%ú"Tú\)	ÚheaderÚ	index_colÚ	na_valuesÚkeep_default_naÚcommentÚ	quotecharÚskipinitialspaceÚ
escapecharr!   zwThe number of columns provided by OpenML does not match the number of columns inferred by pandas when reading the file.z^'(?P<contents>.*)'$c                 óZ   •— t        j                  ‰| «      }|€| S |j                  d«      S )NÚcontents)ÚreÚsearchÚgroup)Úinput_stringÚmatchÚsingle_quote_patterns     €r   Ústrip_single_quotesz0_pandas_arff_parser.<locals>.strip_single_quotes±  s.   ø€ Ü—	‘	Ð.°Ó=ˆØˆ=ØÐà�{‰{˜:Ó&Ð&r1   r:   )r:   r5   rY   Ú
startswithr   Úread_csvr?   r[   ÚerrorsÚParserErrorr˜   ÚcompilerW   ÚitemsrP   ÚCategoricalDtypert   Úrename_categoriesr0   Úto_numpyru   Útolist)r6   rl   rm   rn   ro   Úread_csv_kwargsrv   r7   rW   rs   r€   r   Údtypes_positionalÚdefault_read_csv_kwargsr,   Úexcrr   r}   r~   rž   r!   Úcategorical_columnsr/   r)   ru   r�   s                            @r   Ú_pandas_arff_parserr®   7  sñ  ø€ óp ð ò ˆØ�;‰;�wÓ×%Ñ%Ó'×2Ñ2°7Õ;Ùðð €FØ#ò &ˆØ*¨4Ñ0°Ñ=ˆØ×ÑÓ 9Ò,ð #ˆF�4ŠLØ×ÑÓ! YÓ.Ø%ˆF�4ŠLð&ô 'Ð':Ó;÷áˆG�TØ�6‰>ð 	�˜‘ÑðÐñ ð ØØ�UØ ØØØ ØØ"ñ
Ðð MÐ0ÐL°_Ò5JÈÐL€OØˆB�K‰K˜	Ñ5 _Ñ5€Eð
ð
 +>Ö> $šÒ>ˆŒð 0Ð2HÑHÐØ&+§m¡mÖP˜s°sÐ>OÒ7O’sÐP€OÐPØ�/Ñ"€Eô Ÿ:™:Ð&=Ó>Ðô'ð !Ÿ<™<×-Ñ-Ó/÷áˆD�%Ü�e˜R×0Ñ0Ô1ò 	ðÐñ ð
 #ò KˆØ˜3‘Z—^‘^×5Ñ5Ð6IÓJˆˆcŠ
ðKô ˜uÐ&=Ð?UÓV�D€A€qà˜XÒ%Ø�!�U˜DÐ Ð à�z‰z‹|˜QŸZ™Z›\ˆ1ˆð !Ÿ<™<×-Ñ-Ó/÷áˆD�%Ü�e˜R×0Ñ0Ô1ð 	ˆe×Ñ×%Ñ%Ó'Ñ'ð€Jñ ð
 ˆa��zÐ!Ð!ùóWùò0 ?øÜò Ø�i‰i×#Ñ#ð@ó
ð ð	ûðüò Qùó.ùósH   ÂH-ÃH8 Ã	H3ÃH8 Ã8	I%ÄI%Å"I*Ç);I0È3H8 È8	I"ÉIÉI"c                 ót   — |dk(  rt        | |||||«      S |dk(  rt        | |||||«      S t        d|› d�«      ‚)a6  Load a compressed ARFF file using a given parser.

    Parameters
    ----------
    gzip_file : GzipFile instance
        The file compressed to be read.

    parser : {"pandas", "liac-arff"}
        The parser used to parse the ARFF file. "pandas" is recommended
        but only supports loading dense datasets.

    output_type : {"numpy", "sparse", "pandas"}
        The type of the arrays that will be returned. The possibilities ara:

        - `"numpy"`: both `X` and `y` will be NumPy arrays;
        - `"sparse"`: `X` will be sparse matrix and `y` will be a NumPy array;
        - `"pandas"`: `X` will be a pandas DataFrame and `y` will be either a
          pandas Series or DataFrame.

    openml_columns_info : dict
        The information provided by OpenML regarding the columns of the ARFF
        file.

    feature_names_to_select : list of str
        A list of the feature names to be selected.

    target_names_to_select : list of str
        A list of the target names to be selected.

    read_csv_kwargs : dict, default=None
        Keyword arguments to pass to `pandas.read_csv`. It allows to overwrite
        the default options.

    Returns
    -------
    X : {ndarray, sparse matrix, dataframe}
        The data matrix.

    y : {ndarray, dataframe, series}
        The target.

    frame : dataframe or None
        A dataframe containing both `X` and `y`. `None` if
        `output_array_type != "pandas"`.

    categories : list of str or None
        The names of the features that are categorical. `None` if
        `output_array_type == "pandas"`.
    z	liac-arffr:   zUnknown parser: 'z%'. Should be 'liac-arff' or 'pandas'.)rˆ   r®   r[   )r6   ÚparserÚoutput_typerm   rn   ro   rK   r©   s           r   Úload_arff_from_gzip_filer²   Ï  sq   € ðv �ÒÜ ØØØØ#Ø"Øó
ð 	
ð 
�8Ò	Ü"ØØØØ#Ø"Øó
ð 	
ô Ø ˜xÐ'LÐMó
ð 	
r1   )N)NN)Ú__doc__r]   r˜   Úcollectionsr   Úcollections.abcr   Útypingr   Únumpyr$   Úscipyrb   Ú	externalsr   Úexternals._arffr   Úutils._chunkingr	   r
   Úutils._optional_dependenciesr   Úutils.fixesr   r   Úndarrayr*   r0   rˆ   r®   r²   © r1   r   ú<module>rÀ      s™   ðÙ ?ó
 Û 	Ý #Ý %Ý ã Û å Ý 0ß ?Ý ?Ý #ð Ø!ð Ø48ð àó ðFØ!ðØ48ðà‡Z�Zóò$ðL óI"ðd óU"ð~ ØôP
r1   