i_dZddlmZddlZddlZddlZddlmZmZm Z ddl m Z m Z ddl mZddlmZddlmZmZdd lmZdd lmZdd lmZmZdd lmZdd lmZmZm Z m!Z!m"Z"erddl#m$Z$m%Z%m&Z&m'Z'm(Z(m)Z)d>dZ* d?d@d Z+Gd!dZ,Gd"d#e,Z-Gd$d%e,Z. dAdBd3Z/ed4d&ddej0dddfdCd=Z1dS)Dzparquet compat) annotationsN) TYPE_CHECKINGAnyLiteral)catch_warningsfilterwarnings)lib)import_optional_dependency)AbstractMethodErrorPandas4Warning) set_module)check_dtype_backend) DataFrame get_option)arrow_table_to_pandas) IOHandles get_handle is_fsspec_urlis_urlstringify_path) DtypeBackendFilePathParquetCompressionOptions ReadBufferStorageOptions WriteBufferenginestrreturnBaseImplcd|dkrtd}|dkr_ttg}d}|D]:} |cS#t$r}|dt |zz }Yd}~3d}~wwxYwtd||dkrtS|dkrtSt d ) zreturn our implementationautozio.parquet.enginez - NzUnable to find a usable engine; tried using: 'pyarrow', 'fastparquet'. A suitable version of pyarrow or fastparquet is required for parquet support. Trying to import the above resulted in these errors:pyarrow fastparquetz.engine must be one of 'pyarrow', 'fastparquet')r PyArrowImplFastParquetImpl ImportErrorr ValueError)rengine_classes error_msgs engine_classerrs ?C:\PYTHON\_runtimes\venv\Lib\site-packages\pandas/io/parquet.py get_enginer/4s /00 %7 * 1 1L 1#|~~%%% 1 1 1gC00  1        }} =    E F FFs = A&A!!A&rbFpath1FilePath | ReadBuffer[bytes] | WriteBuffer[bytes]fsrstorage_optionsStorageOptions | Nonemodeis_dirboolVtuple[FilePath | ReadBuffer[bytes] | WriteBuffer[bytes], IOHandles[bytes] | None, Any]c`t|}|tdd}tdd}|'t||jr|rt dnA|t||jjrn$tdt|j t|r||Ttd}td} |j |\}}n#t|j f$rYnwxYw|'td}|jj|fi|pi\}}n&|r$t!|r|d krtd d} |sR|sPt|t"r;t$j|st+||d | } d}| j}|| |fS) zFile handling for PyArrow.Nz pyarrow.fsignore)errorsfsspecz8storage_options not supported with a pyarrow FileSystem.z9filesystem must be a pyarrow or fsspec FileSystem, not a r$r0z8storage_options passed with buffer, or non-supported URLFis_textr4)rr isinstance FileSystemNotImplementedErrorspecAbstractFileSystemr)type__name__rfrom_uri TypeError ArrowInvalidcore url_to_fsrrosr1isdirrhandle) r1r3r4r6r7path_or_handlepa_fsr=pahandless r._get_path_or_handlerSVs2$D))N ~*<III+HXFFF  B0@!A!A  )N  Jr6;3Q$R$R  -b*-- ^$$U  "+I66B.|<%>t%D%D"NNr/     :/99F!6!6""#2#8b"" B U&"8"8UDDLLSTTTG  ( ( ~s + + ( n-- ( D%     7B &&sC..DDc8eZdZed dZd dZd d dZdS) r dfrrNonecNt|tstddS)Nz+to_parquet only supports IO with DataFrames)r@rr))rUs r.validate_dataframezBaseImpl.validate_dataframes0"i(( LJKK K L Lc  t|Nr )selfrUr1 compressionkwargss r.writezBaseImpl.write!$'''rYNc  t|r[r\)r]r1columnsr_s r.readz BaseImpl.readrarY)rUrrrVr[)rr)rF __module__ __qualname__ staticmethodrXr`rdrYr.r r scLLL\L(((((((((((rYcJeZdZddZ dddZddejdddfddZdS)r&rrVcFtddddl}ddl}||_dS)Nr$z(pyarrow is required for parquet support.extrar)r pyarrow.parquet(pandas.core.arrays.arrow.extension_typesapi)r]r$pandass r.__init__zPyArrowImpl.__init__sF" G      8777rYsnappyNrUrr1FilePath | WriteBuffer[bytes]r^rindex bool | Noner4r5partition_colslist[str] | Nonec R||d|ddi} ||| d<|jjj|fi| } |jrBdt j|ji} | jj } i| | } | | } t|||d|du\}}}t|tjrlt|dr\t|jt"t$fr;t|jt$r|j}n|j} ||jjj| |f|||d|n|jjj| |f||d|||dSdS#||wwxYw) Nschemapreserve_index PANDAS_ATTRSwb)r4r6r7name)r^rv filesystem)r^r~)rXpoproTable from_pandasattrsjsondumpsrymetadatareplace_schema_metadatarSr@ioBufferedWriterhasattrr}rbytesdecodeparquetwrite_to_dataset write_tableclose)r]rUr1r^rtr4rvr~r_from_pandas_kwargstable df_metadataexisting_metadatamerged_metadatarOrRs r.r`zPyArrowImpl.writes( ###.6 8T8R8R-S  38 / 0**2DD1CDD 8 C)4:bh+?+?@K % 5 B!2BkBO11/BBE.A  +!- / / / + ~r'8 9 9 5// 5>.e == 5 .-u55 5!/!4!;!;!=!=!/!4 )1 1"!,#1)   - ,"!,)   " #"w" #s 7>>"*/":?"KK#':k#:#:FL" #w" #s0*C&)A=1 C&=BC&BA C&&C?rrVrrNNNN)rUrr1rsr^rrtrur4r5rvrwrrV)rrr4r5rrrr)rFrerfrqr`r no_defaultrdrhrYr.r&r&s    2:!15+/@ @ @ @ @ J69n1526. . . . . . . rYr&c>eZdZddZ ddd Z dddZdS)r'rrVc6tdd}||_dS)Nr%z,fastparquet is required for parquet support.rk)r ro)r]r%s r.rqzFastParquetImpl.__init__"s+1 !O   rYrrNrUrr^*Literal['snappy', 'gzip', 'brotli'] | Noner4r5c  ||d|vr|tdd|vr|d}|d|d<|tdt |}t |rt d fd|d<nrtd td 5|jj ||f|||d |ddddS#1swxYwYdS) N partition_onzYCannot use both partition_on and partition_cols. Use partition_cols for partitioning datahive file_scheme9filesystem is not implemented for the fastparquet engine.r=cJj|dfipiS)Nr|)open)r1_r=r4s r.z'FastParquetImpl.write..Ms8+&+d33.4"33dffrY open_withz?storage_options passed with file object or non-fsspec file pathT)record)r^ write_indexr) rXr)rrBrrr rror`) r]rUr1r^rtrvr4r~r_r=s ` @r.r`zFastParquetImpl.write*s ### V # #(BK  V # ##ZZ77N  %$*F= !  !%K  d##    /99F#####F;   Q 4 ( ( (   DHN (!+                          s6CC #C r dict | Nonec Xi}|dtj} d|d<| tjurtd|t d|t dt |}d} t |r)td} | j|dfi|pij |d <nNt|tr9tj |st|dd| } | j} |jj|fi|} t'5t)d d t*| jd||d |cddd| | SS#1swxYwY | | dSdS#| | wwxYw)NrF pandas_nullszHThe 'dtype_backend' argument is not supported for the fastparquet enginerz?to_pandas_kwargs is not implemented for the fastparquet engine.r=r0r3r>r;r)rcrrh)rr rr)rBrrr rr3r@rrLr1rMrrNro ParquetFilerrr to_pandasr) r]r1rcrr4r~rr_parquet_kwargsrrRr= parquet_files r.rdzFastParquetImpl.read_sY*, ?CNCC ).~&  . .%   !%K   '%Q d##    "/99F#.6;tT#U#Uo>SQS#U#U#XN4 c " " "27==+>+> "!dE?G>D /48/GGGGL!!  ." .|-#W8>        " #         " #"w" #s0?!F &E* F*E..F1E.2FF)rr)rUrr^rr4r5rrV)NNNNN)r4r5rrrr)rFrerfrqr`rdrhrYr.r'r'!sCK1533333p15(,7 7 7 7 7 7 7 rYr'r"rrrUr$FilePath | WriteBuffer[bytes] | Noner^rrtrurvrwr~ bytes | Nonec t|tr|g}t|} |tjn|} | j|| f|||||d||0t| tjsJ| SdS)a Write a DataFrame to the parquet format. Parameters ---------- df : DataFrame path : str, path object, file-like object, or None, default None String, path object (implementing ``os.PathLike[str]``), or file-like object implementing a binary ``write()`` function. If None, the result is returned as bytes. If a string, it will be used as Root Directory path when writing a partitioned dataset. The engine fastparquet does not accept file-like objects. engine : {'auto', 'pyarrow', 'fastparquet'}, default 'auto' Parquet library to use. If 'auto', then the option ``io.parquet.engine`` is used. The default ``io.parquet.engine`` behavior is to try 'pyarrow', falling back to 'fastparquet' if 'pyarrow' is unavailable. When using the ``'pyarrow'`` engine and no storage options are provided and a filesystem is implemented by both ``pyarrow.fs`` and ``fsspec`` (e.g. "s3://"), then the ``pyarrow.fs`` filesystem is attempted first. Use the filesystem keyword with an instantiated fsspec filesystem if you wish to use its implementation. compression : {'snappy', 'gzip', 'brotli', 'lz4', 'zstd', None}, default 'snappy'. Name of the compression to use. Use ``None`` for no compression. index : bool, default None If ``True``, include the dataframe's index(es) in the file output. If ``False``, they will not be written to the file. If ``None``, similar to ``True`` the dataframe's index(es) will be saved. However, instead of being saved as values, the RangeIndex will be stored as a range in the metadata so it doesn't require much space and is faster. Other indexes will be included as columns in the file output. partition_cols : str or list, optional, default None Column names by which to partition the dataset. Columns are partitioned in the order they are given. Must be None if path is not a string. storage_options : dict, optional Extra options that make sense for a particular storage connection, e.g. host, port, username, password, etc. For HTTP(S) URLs the key-value pairs are forwarded to ``urllib.request.Request`` as header options. For other URLs (e.g. starting with "s3://", and "gcs://") the key-value pairs are forwarded to ``fsspec.open``. Please see ``fsspec`` and ``urllib`` for more details, and for more examples on storage options refer `here `_. filesystem : fsspec or pyarrow filesystem, default None Filesystem object to use when reading the parquet file. Only implemented for ``engine="pyarrow"``. .. versionadded:: 2.1.0 **kwargs Additional keyword arguments passed to the engine: * For ``engine="pyarrow"``: passed to :func:`pyarrow.parquet.write_table` or :func:`pyarrow.parquet.write_to_dataset` (when using partition_cols) * For ``engine="fastparquet"``: passed to :func:`fastparquet.write` Returns ------- bytes if no path argument is provided else None N)r^rtrvr4r~)r@rr/rBytesIOr`getvalue) rUr1rr^rtr4rvr~r_impl path_or_bufs r. to_parquetrsV.#&&*() f  DAESWKDJ   %'       |+rz22222##%%%trYrpFilePath | ReadBuffer[bytes]rcrrr&list[tuple] | list[list[tuple]] | Nonerrc ht|} t|| j|f||||||d|S)aE Load a parquet object from the file path, returning a DataFrame. The function automatically handles reading the data from a parquet file and creates a DataFrame with the appropriate structure. Parameters ---------- path : str, path object or file-like object String, path object (implementing ``os.PathLike[str]``), or file-like object implementing a binary ``read()`` function. The string could be a URL. Valid URL schemes include http, ftp, s3, gs, and file. For file URLs, a host is expected. A local file could be: ``file://localhost/path/to/table.parquet``. A file URL can also be a path to a directory that contains multiple partitioned parquet files. Both pyarrow and fastparquet support paths to directories as well as file URLs. A directory path could be: ``file://localhost/path/to/tables`` or ``s3://bucket/partition_dir``. engine : {'auto', 'pyarrow', 'fastparquet'}, default 'auto' Parquet library to use. If 'auto', then the option ``io.parquet.engine`` is used. The default ``io.parquet.engine`` behavior is to try 'pyarrow', falling back to 'fastparquet' if 'pyarrow' is unavailable. When using the ``'pyarrow'`` engine and no storage options are provided and a filesystem is implemented by both ``pyarrow.fs`` and ``fsspec`` (e.g. "s3://"), then the ``pyarrow.fs`` filesystem is attempted first. Use the filesystem keyword with an instantiated fsspec filesystem if you wish to use its implementation. columns : list, default=None If not None, only these columns will be read from the file. storage_options : dict, optional Extra options that make sense for a particular storage connection, e.g. host, port, username, password, etc. For HTTP(S) URLs the key-value pairs are forwarded to ``urllib.request.Request`` as header options. For other URLs (e.g. starting with "s3://", and "gcs://") the key-value pairs are forwarded to ``fsspec.open``. Please see ``fsspec`` and ``urllib`` for more details, and for more examples on storage options refer `here `_. dtype_backend : {'numpy_nullable', 'pyarrow'} Back-end data type applied to the resultant :class:`DataFrame` (still experimental). If not specified, the default behavior is to not use nullable data types. If specified, the behavior is as follows: * ``"numpy_nullable"``: returns nullable-dtype-backed :class:`DataFrame` * ``"pyarrow"``: returns pyarrow-backed nullable :class:`ArrowDtype` :class:`DataFrame` .. versionadded:: 2.0 filesystem : fsspec or pyarrow filesystem, default None Filesystem object to use when reading the parquet file. Only implemented for ``engine="pyarrow"``. .. versionadded:: 2.1.0 filters : List[Tuple] or List[List[Tuple]], default None To filter out data. Filter syntax: [[(column, op, val), ...],...] where op is [==, =, >, >=, <, <=, !=, in, not in] The innermost tuples are transposed into a set of filters applied through an `AND` operation. The outer list combines these sets of filters through an `OR` operation. A single list of tuples can also be used, meaning that no `OR` operation between set of filters is to be conducted. Using this argument will NOT result in row-wise filtering of the final partitions unless ``engine="pyarrow"`` is also specified. For other engines, filtering is only performed at the partition level, that is, to prevent the loading of some row-groups and/or files. .. versionadded:: 2.1.0 to_pandas_kwargs : dict | None, default None Keyword arguments to pass through to :func:`pyarrow.Table.to_pandas` when ``engine="pyarrow"``. .. versionadded:: 3.0.0 **kwargs Additional keyword arguments passed to the engine: * For ``engine="pyarrow"``: passed to :func:`pyarrow.parquet.read_table` * For ``engine="fastparquet"``: passed to :meth:`fastparquet.ParquetFile.to_pandas` Returns ------- DataFrame DataFrame based on parquet file. See Also -------- DataFrame.to_parquet : Create a parquet object that serializes a DataFrame. Examples -------- >>> original_df = pd.DataFrame({"foo": range(5), "bar": range(5, 10)}) >>> original_df foo bar 0 0 5 1 1 6 2 2 7 3 3 8 4 4 9 >>> df_parquet_bytes = original_df.to_parquet() >>> from io import BytesIO >>> restored_df = pd.read_parquet(BytesIO(df_parquet_bytes)) >>> restored_df foo bar 0 0 5 1 1 6 2 2 7 3 3 8 4 4 9 >>> restored_df.equals(original_df) True >>> restored_bar = pd.read_parquet(BytesIO(df_parquet_bytes), columns=["bar"]) >>> restored_bar bar 0 5 1 6 2 7 3 8 4 9 >>> restored_bar.equals(original_df[["bar"]]) True The function uses `kwargs` that are passed directly to the engine. In the following example, we use the `filters` argument of the pyarrow engine to filter the rows of the DataFrame. Since `pyarrow` is the default engine, we can omit the `engine` argument. Note that the `filters` argument is implemented by the `pyarrow` engine, which can benefit from multithreading and also potentially be more economical in terms of memory. >>> sel = [("foo", ">", 2)] >>> restored_part = pd.read_parquet(BytesIO(df_parquet_bytes), filters=sel) >>> restored_part foo bar 0 3 8 1 4 9 )rcrr4rr~r)r/rrd) r1rrcr4rr~rrr_rs r. read_parquetrs_@ f  D &&& 49  '#)      rY)rrrr )Nr0F) r1r2r3rr4r5r6rr7r8rr9)Nr"rrNNNN)rUrr1rrrr^rrtrur4r5rvrwr~rrr)r1rrrrcrwr4r5rrr~rrrrrrr)2__doc__ __future__rrrrLtypingrrrwarningsrr pandas._libsr pandas.compat._optionalr pandas.errorsr r pandas.util._decoratorsr pandas.util._validatorsrrprrpandas.io._utilrpandas.io.commonrrrrrpandas._typingrrrrrrr/rSr r&r'rrrrhrYr.rs>""""""   >>>>>>/.....777777 211111GGGGJ.2 <'<'<'<'<'~ ( ( ( ( ( ( ( (| | | | | (| | | ~u u u u u hu u u t26-5-1'+`````F H $-125.6:$(kkkkkkkrY