Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions docs/source/python/api/ipc.rst
Original file line number Diff line number Diff line change
Expand Up @@ -28,6 +28,10 @@ Inter-Process Communication
.. autosummary::
:toctree: ../generated/

ipc.write_file
ipc.read_file
ipc.read_feather
ipc.FileDataset
ipc.new_file
ipc.open_file
ipc.new_stream
Expand Down
23 changes: 8 additions & 15 deletions docs/source/python/feather.rst
Original file line number Diff line number Diff line change
Expand Up @@ -115,8 +115,8 @@ intend to maintain read support for V1 for the foreseeable future.
Migration to IPC
----------------

Since Feather V2 is the Arrow IPC file format, you can use the
:mod:`pyarrow.ipc` module as a direct replacement:
Since Feather V2 is the Arrow IPC file format, use the high-level functions in
:mod:`pyarrow.ipc` as direct replacements:

.. code-block:: python

Expand All @@ -126,20 +126,13 @@ Since Feather V2 is the Arrow IPC file format, you can use the
table = pa.table({"col1": [1, 2, 3], "col2": ["a", "b", "c"]})

# Writing (replaces feather.write_feather)
options = pa.ipc.IpcWriteOptions(compression='lz4')
with pa.ipc.new_file("data.arrow", table.schema, options=options) as writer:
writer.write_table(table)
pa.ipc.write_file(table, "data.arrow")

# Reading (replaces feather.read_table)
with pa.ipc.open_file("data.arrow") as reader:
result = reader.read_all()
result = pa.ipc.read_file("data.arrow")

.. note::
# Reading as pandas (replaces feather.read_feather)
dataframe = pa.ipc.read_feather("data.arrow")

``feather.write_feather`` defaults to LZ4 compression, while
``ipc.new_file`` does not compress by default. To preserve the same
behavior, pass ``compression='lz4'`` via
:class:`~pyarrow.ipc.IpcWriteOptions` as shown above.

For reading multiple files, use the :mod:`pyarrow.dataset` module with
``format='ipc'`` instead of :class:`~pyarrow.feather.FeatherDataset`.
For reading multiple files, use :class:`pyarrow.ipc.FileDataset` or the
:mod:`pyarrow.dataset` module with ``format='ipc'``.
110 changes: 65 additions & 45 deletions python/pyarrow/feather.py
Original file line number Diff line number Diff line change
Expand Up @@ -28,12 +28,9 @@
from pyarrow._feather import FeatherError # noqa: F401


class FeatherDataset:
class _FileDataset:
"""
Encapsulates details of reading a list of Feather files.

.. deprecated:: 24.0.0
Use :func:`pyarrow.dataset.dataset` with ``format='ipc'`` instead.
Encapsulates details of reading a list of Arrow IPC files.

Parameters
----------
Expand All @@ -44,12 +41,6 @@ class FeatherDataset:
"""

def __init__(self, path_or_paths, validate_schema=True):
warnings.warn(
"pyarrow.feather.FeatherDataset is deprecated as of 24.0.0. "
"Use pyarrow.dataset.dataset() with format='ipc' instead.",
FutureWarning,
stacklevel=2
)
self.paths = path_or_paths
self.validate_schema = validate_schema

Expand Down Expand Up @@ -103,6 +94,24 @@ def read_pandas(self, columns=None, use_threads=True):
use_threads=use_threads)


class FeatherDataset(_FileDataset):
"""
Deprecated alias for reading a list of Feather files.

.. deprecated:: 24.0.0
Use :class:`pyarrow.ipc.FileDataset` instead.
"""

def __init__(self, path_or_paths, validate_schema=True):
warnings.warn(
"pyarrow.feather.FeatherDataset is deprecated as of 24.0.0. "
"Use pyarrow.ipc.FileDataset instead.",
DeprecationWarning,
stacklevel=2
)
super().__init__(path_or_paths, validate_schema=validate_schema)


def check_chunked_overflow(name, col):
if col.num_chunks == 1:
return
Expand All @@ -122,15 +131,10 @@ def check_chunked_overflow(name, col):
_FEATHER_SUPPORTED_CODECS = {'lz4', 'zstd', 'uncompressed'}


def write_feather(df, dest, compression=None, compression_level=None,
chunksize=None, version=2):
def _write_file(df, dest, compression=None, compression_level=None,
chunksize=None, version=2):
"""
Write a pandas.DataFrame to Feather format.

.. deprecated:: 24.0.0
Use :func:`pyarrow.ipc.new_file` /
:class:`pyarrow.ipc.RecordBatchFileWriter` instead.
Feather V2 is the Arrow IPC file format.
Write a pandas.DataFrame to the Arrow IPC or legacy Feather format.

Parameters
----------
Expand All @@ -152,13 +156,6 @@ def write_feather(df, dest, compression=None, compression_level=None,
Feather file version. Version 2 is the current. Version 1 is the more
limited legacy format
"""
warnings.warn(
"pyarrow.feather.write_feather is deprecated as of 24.0.0. "
"Use pyarrow.ipc.new_file() / RecordBatchFileWriter instead. "
"Feather V2 is the Arrow IPC file format.",
FutureWarning,
stacklevel=2
)
if _pandas_api.have_pandas:
if (_pandas_api.has_sparse and
isinstance(df, _pandas_api.pd.SparseDataFrame)):
Expand Down Expand Up @@ -217,16 +214,30 @@ def write_feather(df, dest, compression=None, compression_level=None,
raise


def read_feather(source, columns=None, use_threads=True,
memory_map=False, **kwargs):
def write_feather(df, dest, compression=None, compression_level=None,
chunksize=None, version=2):
"""
Read a pandas.DataFrame from Feather format. To read as pyarrow.Table use
feather.read_table.
Deprecated alias for writing Feather files.

.. deprecated:: 24.0.0
Use :func:`pyarrow.ipc.open_file` /
:class:`pyarrow.ipc.RecordBatchFileReader` instead.
Feather V2 is the Arrow IPC file format.
Use :func:`pyarrow.ipc.write_file` instead.
"""
warnings.warn(
"pyarrow.feather.write_feather is deprecated as of 24.0.0. "
"Use pyarrow.ipc.write_file instead.",
DeprecationWarning,
stacklevel=2
)
return _write_file(
df, dest, compression=compression,
compression_level=compression_level, chunksize=chunksize,
version=version)


def _read_pandas(source, columns=None, use_threads=True,
memory_map=False, **kwargs):
"""
Read an Arrow IPC or legacy Feather file as a pandas.DataFrame.

Parameters
----------
Expand All @@ -249,16 +260,28 @@ def read_feather(source, columns=None, use_threads=True,
df : pandas.DataFrame
The contents of the Feather file as a pandas.DataFrame
"""
return (_read_table_internal(
source, columns=columns, memory_map=memory_map,
use_threads=use_threads).to_pandas(use_threads=use_threads, **kwargs))


def read_feather(source, columns=None, use_threads=True,
memory_map=False, **kwargs):
"""
Deprecated alias for reading Feather files as a pandas.DataFrame.

.. deprecated:: 24.0.0
Use :func:`pyarrow.ipc.read_feather` instead.
"""
warnings.warn(
"pyarrow.feather.read_feather is deprecated as of 24.0.0. "
"Use pyarrow.ipc.open_file() / RecordBatchFileReader instead. "
"Feather V2 is the Arrow IPC file format.",
FutureWarning,
"Use pyarrow.ipc.read_feather instead.",
DeprecationWarning,
stacklevel=2
)
return (_read_table_internal(
source, columns=columns, memory_map=memory_map,
use_threads=use_threads).to_pandas(use_threads=use_threads, **kwargs))
return _read_pandas(
source, columns=columns, use_threads=use_threads,
memory_map=memory_map, **kwargs)


def _read_table_internal(source, columns=None, memory_map=False,
Expand Down Expand Up @@ -303,9 +326,7 @@ def read_table(source, columns=None, memory_map=False, use_threads=True):
Read a pyarrow.Table from Feather format

.. deprecated:: 24.0.0
Use :func:`pyarrow.ipc.open_file` /
:class:`pyarrow.ipc.RecordBatchFileReader` instead.
Feather V2 is the Arrow IPC file format.
Use :func:`pyarrow.ipc.read_file` instead.

Parameters
----------
Expand All @@ -326,9 +347,8 @@ def read_table(source, columns=None, memory_map=False, use_threads=True):
"""
warnings.warn(
"pyarrow.feather.read_table is deprecated as of 24.0.0. "
"Use pyarrow.ipc.open_file() / RecordBatchFileReader instead. "
"Feather V2 is the Arrow IPC file format.",
FutureWarning,
"Use pyarrow.ipc.read_file instead.",
DeprecationWarning,
stacklevel=2
)
return _read_table_internal(source, columns=columns,
Expand Down
6 changes: 6 additions & 0 deletions python/pyarrow/ipc.py
Original file line number Diff line number Diff line change
Expand Up @@ -29,6 +29,12 @@
read_tensor, write_tensor,
get_record_batch_size, get_tensor_size)
import pyarrow.lib as lib
from pyarrow.feather import (
_FileDataset as FileDataset,
_read_pandas as read_feather,
_read_table_internal as read_file,
_write_file as write_file,
)


class RecordBatchStreamReader(lib._RecordBatchStreamReader):
Expand Down
Loading