From d756630038a5502a58810b0010d234f5b26e5267 Mon Sep 17 00:00:00 2001 From: Rok Mihevc Date: Tue, 4 Aug 2026 01:56:25 +0200 Subject: [PATCH 1/3] GH-49232: [Python] Add IPC aliases for Feather APIs --- docs/source/python/api/ipc.rst | 4 ++ python/pyarrow/feather.py | 99 +++++++++++++++++++++------------- python/pyarrow/ipc.py | 6 +++ 3 files changed, 73 insertions(+), 36 deletions(-) diff --git a/docs/source/python/api/ipc.rst b/docs/source/python/api/ipc.rst index 027fee583ec1..55cfe14bd2e3 100644 --- a/docs/source/python/api/ipc.rst +++ b/docs/source/python/api/ipc.rst @@ -28,6 +28,10 @@ Inter-Process Communication .. autosummary:: :toctree: ../generated/ + ipc.write_file + ipc.read_file + ipc.read_feather + ipc.FileDataset ipc.new_file ipc.open_file ipc.new_stream diff --git a/python/pyarrow/feather.py b/python/pyarrow/feather.py index 68f708c91489..db505fe8e38f 100644 --- a/python/pyarrow/feather.py +++ b/python/pyarrow/feather.py @@ -28,12 +28,9 @@ from pyarrow._feather import FeatherError # noqa: F401 -class FeatherDataset: +class _FileDataset: """ - Encapsulates details of reading a list of Feather files. - - .. deprecated:: 24.0.0 - Use :func:`pyarrow.dataset.dataset` with ``format='ipc'`` instead. + Encapsulates details of reading a list of Arrow IPC files. Parameters ---------- @@ -44,12 +41,6 @@ class FeatherDataset: """ def __init__(self, path_or_paths, validate_schema=True): - warnings.warn( - "pyarrow.feather.FeatherDataset is deprecated as of 24.0.0. " - "Use pyarrow.dataset.dataset() with format='ipc' instead.", - FutureWarning, - stacklevel=2 - ) self.paths = path_or_paths self.validate_schema = validate_schema @@ -103,6 +94,24 @@ def read_pandas(self, columns=None, use_threads=True): use_threads=use_threads) +class FeatherDataset(_FileDataset): + """ + Deprecated alias for reading a list of Feather files. + + .. deprecated:: 24.0.0 + Use :func:`pyarrow.dataset.dataset` with ``format='ipc'`` instead. + """ + + def __init__(self, path_or_paths, validate_schema=True): + warnings.warn( + "pyarrow.feather.FeatherDataset is deprecated as of 24.0.0. " + "Use pyarrow.dataset.dataset() with format='ipc' instead.", + FutureWarning, + stacklevel=2 + ) + super().__init__(path_or_paths, validate_schema=validate_schema) + + def check_chunked_overflow(name, col): if col.num_chunks == 1: return @@ -122,15 +131,10 @@ def check_chunked_overflow(name, col): _FEATHER_SUPPORTED_CODECS = {'lz4', 'zstd', 'uncompressed'} -def write_feather(df, dest, compression=None, compression_level=None, - chunksize=None, version=2): +def _write_file(df, dest, compression=None, compression_level=None, + chunksize=None, version=2): """ - Write a pandas.DataFrame to Feather format. - - .. deprecated:: 24.0.0 - Use :func:`pyarrow.ipc.new_file` / - :class:`pyarrow.ipc.RecordBatchFileWriter` instead. - Feather V2 is the Arrow IPC file format. + Write a pandas.DataFrame to the Arrow IPC or legacy Feather format. Parameters ---------- @@ -152,13 +156,6 @@ def write_feather(df, dest, compression=None, compression_level=None, Feather file version. Version 2 is the current. Version 1 is the more limited legacy format """ - warnings.warn( - "pyarrow.feather.write_feather is deprecated as of 24.0.0. " - "Use pyarrow.ipc.new_file() / RecordBatchFileWriter instead. " - "Feather V2 is the Arrow IPC file format.", - FutureWarning, - stacklevel=2 - ) if _pandas_api.have_pandas: if (_pandas_api.has_sparse and isinstance(df, _pandas_api.pd.SparseDataFrame)): @@ -217,16 +214,32 @@ def write_feather(df, dest, compression=None, compression_level=None, raise -def read_feather(source, columns=None, use_threads=True, - memory_map=False, **kwargs): +def write_feather(df, dest, compression=None, compression_level=None, + chunksize=None, version=2): """ - Read a pandas.DataFrame from Feather format. To read as pyarrow.Table use - feather.read_table. + Deprecated alias for writing Feather files. .. deprecated:: 24.0.0 - Use :func:`pyarrow.ipc.open_file` / - :class:`pyarrow.ipc.RecordBatchFileReader` instead. - Feather V2 is the Arrow IPC file format. + Use :func:`pyarrow.ipc.new_file` / + :class:`pyarrow.ipc.RecordBatchFileWriter` instead. + """ + warnings.warn( + "pyarrow.feather.write_feather is deprecated as of 24.0.0. " + "Use pyarrow.ipc.new_file() / RecordBatchFileWriter instead. " + "Feather V2 is the Arrow IPC file format.", + FutureWarning, + stacklevel=2 + ) + return _write_file( + df, dest, compression=compression, + compression_level=compression_level, chunksize=chunksize, + version=version) + + +def _read_pandas(source, columns=None, use_threads=True, + memory_map=False, **kwargs): + """ + Read an Arrow IPC or legacy Feather file as a pandas.DataFrame. Parameters ---------- @@ -249,6 +262,20 @@ def read_feather(source, columns=None, use_threads=True, df : pandas.DataFrame The contents of the Feather file as a pandas.DataFrame """ + return (_read_table_internal( + source, columns=columns, memory_map=memory_map, + use_threads=use_threads).to_pandas(use_threads=use_threads, **kwargs)) + + +def read_feather(source, columns=None, use_threads=True, + memory_map=False, **kwargs): + """ + Deprecated alias for reading Feather files as a pandas.DataFrame. + + .. deprecated:: 24.0.0 + Use :func:`pyarrow.ipc.open_file` / + :class:`pyarrow.ipc.RecordBatchFileReader` instead. + """ warnings.warn( "pyarrow.feather.read_feather is deprecated as of 24.0.0. " "Use pyarrow.ipc.open_file() / RecordBatchFileReader instead. " @@ -256,9 +283,9 @@ def read_feather(source, columns=None, use_threads=True, FutureWarning, stacklevel=2 ) - return (_read_table_internal( - source, columns=columns, memory_map=memory_map, - use_threads=use_threads).to_pandas(use_threads=use_threads, **kwargs)) + return _read_pandas( + source, columns=columns, use_threads=use_threads, + memory_map=memory_map, **kwargs) def _read_table_internal(source, columns=None, memory_map=False, diff --git a/python/pyarrow/ipc.py b/python/pyarrow/ipc.py index 4e236678788a..286838ce0741 100644 --- a/python/pyarrow/ipc.py +++ b/python/pyarrow/ipc.py @@ -29,6 +29,12 @@ read_tensor, write_tensor, get_record_batch_size, get_tensor_size) import pyarrow.lib as lib +from pyarrow.feather import ( + _FileDataset as FileDataset, + _read_pandas as read_feather, + _read_table_internal as read_file, + _write_file as write_file, +) class RecordBatchStreamReader(lib._RecordBatchStreamReader): From c63fd7e06cb82e15c1fc2b8a8b34c66b3569737b Mon Sep 17 00:00:00 2001 From: Rok Mihevc Date: Tue, 4 Aug 2026 01:56:43 +0200 Subject: [PATCH 2/3] GH-49232: [Python] Point Feather users to IPC aliases --- docs/source/python/feather.rst | 23 ++++++++--------------- python/pyarrow/feather.py | 23 ++++++++--------------- 2 files changed, 16 insertions(+), 30 deletions(-) diff --git a/docs/source/python/feather.rst b/docs/source/python/feather.rst index 76520e912b67..c23e3602d166 100644 --- a/docs/source/python/feather.rst +++ b/docs/source/python/feather.rst @@ -115,8 +115,8 @@ intend to maintain read support for V1 for the foreseeable future. Migration to IPC ---------------- -Since Feather V2 is the Arrow IPC file format, you can use the -:mod:`pyarrow.ipc` module as a direct replacement: +Since Feather V2 is the Arrow IPC file format, use the high-level functions in +:mod:`pyarrow.ipc` as direct replacements: .. code-block:: python @@ -126,20 +126,13 @@ Since Feather V2 is the Arrow IPC file format, you can use the table = pa.table({"col1": [1, 2, 3], "col2": ["a", "b", "c"]}) # Writing (replaces feather.write_feather) - options = pa.ipc.IpcWriteOptions(compression='lz4') - with pa.ipc.new_file("data.arrow", table.schema, options=options) as writer: - writer.write_table(table) + pa.ipc.write_file(table, "data.arrow") # Reading (replaces feather.read_table) - with pa.ipc.open_file("data.arrow") as reader: - result = reader.read_all() + result = pa.ipc.read_file("data.arrow") -.. note:: + # Reading as pandas (replaces feather.read_feather) + dataframe = pa.ipc.read_feather("data.arrow") - ``feather.write_feather`` defaults to LZ4 compression, while - ``ipc.new_file`` does not compress by default. To preserve the same - behavior, pass ``compression='lz4'`` via - :class:`~pyarrow.ipc.IpcWriteOptions` as shown above. - -For reading multiple files, use the :mod:`pyarrow.dataset` module with -``format='ipc'`` instead of :class:`~pyarrow.feather.FeatherDataset`. +For reading multiple files, use :class:`pyarrow.ipc.FileDataset` or the +:mod:`pyarrow.dataset` module with ``format='ipc'``. diff --git a/python/pyarrow/feather.py b/python/pyarrow/feather.py index db505fe8e38f..05d232d7ffda 100644 --- a/python/pyarrow/feather.py +++ b/python/pyarrow/feather.py @@ -99,13 +99,13 @@ class FeatherDataset(_FileDataset): Deprecated alias for reading a list of Feather files. .. deprecated:: 24.0.0 - Use :func:`pyarrow.dataset.dataset` with ``format='ipc'`` instead. + Use :class:`pyarrow.ipc.FileDataset` instead. """ def __init__(self, path_or_paths, validate_schema=True): warnings.warn( "pyarrow.feather.FeatherDataset is deprecated as of 24.0.0. " - "Use pyarrow.dataset.dataset() with format='ipc' instead.", + "Use pyarrow.ipc.FileDataset instead.", FutureWarning, stacklevel=2 ) @@ -220,13 +220,11 @@ def write_feather(df, dest, compression=None, compression_level=None, Deprecated alias for writing Feather files. .. deprecated:: 24.0.0 - Use :func:`pyarrow.ipc.new_file` / - :class:`pyarrow.ipc.RecordBatchFileWriter` instead. + Use :func:`pyarrow.ipc.write_file` instead. """ warnings.warn( "pyarrow.feather.write_feather is deprecated as of 24.0.0. " - "Use pyarrow.ipc.new_file() / RecordBatchFileWriter instead. " - "Feather V2 is the Arrow IPC file format.", + "Use pyarrow.ipc.write_file instead.", FutureWarning, stacklevel=2 ) @@ -273,13 +271,11 @@ def read_feather(source, columns=None, use_threads=True, Deprecated alias for reading Feather files as a pandas.DataFrame. .. deprecated:: 24.0.0 - Use :func:`pyarrow.ipc.open_file` / - :class:`pyarrow.ipc.RecordBatchFileReader` instead. + Use :func:`pyarrow.ipc.read_feather` instead. """ warnings.warn( "pyarrow.feather.read_feather is deprecated as of 24.0.0. " - "Use pyarrow.ipc.open_file() / RecordBatchFileReader instead. " - "Feather V2 is the Arrow IPC file format.", + "Use pyarrow.ipc.read_feather instead.", FutureWarning, stacklevel=2 ) @@ -330,9 +326,7 @@ def read_table(source, columns=None, memory_map=False, use_threads=True): Read a pyarrow.Table from Feather format .. deprecated:: 24.0.0 - Use :func:`pyarrow.ipc.open_file` / - :class:`pyarrow.ipc.RecordBatchFileReader` instead. - Feather V2 is the Arrow IPC file format. + Use :func:`pyarrow.ipc.read_file` instead. Parameters ---------- @@ -353,8 +347,7 @@ def read_table(source, columns=None, memory_map=False, use_threads=True): """ warnings.warn( "pyarrow.feather.read_table is deprecated as of 24.0.0. " - "Use pyarrow.ipc.open_file() / RecordBatchFileReader instead. " - "Feather V2 is the Arrow IPC file format.", + "Use pyarrow.ipc.read_file instead.", FutureWarning, stacklevel=2 ) From 6f8f7862cea8b4bb52d9be768a37f5241d7cd5b7 Mon Sep 17 00:00:00 2001 From: Rok Mihevc Date: Tue, 4 Aug 2026 01:50:58 +0200 Subject: [PATCH 3/3] GH-49232: [Python] Use DeprecationWarning for Feather APIs --- python/pyarrow/feather.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/python/pyarrow/feather.py b/python/pyarrow/feather.py index 05d232d7ffda..e83c953b2097 100644 --- a/python/pyarrow/feather.py +++ b/python/pyarrow/feather.py @@ -106,7 +106,7 @@ def __init__(self, path_or_paths, validate_schema=True): warnings.warn( "pyarrow.feather.FeatherDataset is deprecated as of 24.0.0. " "Use pyarrow.ipc.FileDataset instead.", - FutureWarning, + DeprecationWarning, stacklevel=2 ) super().__init__(path_or_paths, validate_schema=validate_schema) @@ -225,7 +225,7 @@ def write_feather(df, dest, compression=None, compression_level=None, warnings.warn( "pyarrow.feather.write_feather is deprecated as of 24.0.0. " "Use pyarrow.ipc.write_file instead.", - FutureWarning, + DeprecationWarning, stacklevel=2 ) return _write_file( @@ -276,7 +276,7 @@ def read_feather(source, columns=None, use_threads=True, warnings.warn( "pyarrow.feather.read_feather is deprecated as of 24.0.0. " "Use pyarrow.ipc.read_feather instead.", - FutureWarning, + DeprecationWarning, stacklevel=2 ) return _read_pandas( @@ -348,7 +348,7 @@ def read_table(source, columns=None, memory_map=False, use_threads=True): warnings.warn( "pyarrow.feather.read_table is deprecated as of 24.0.0. " "Use pyarrow.ipc.read_file instead.", - FutureWarning, + DeprecationWarning, stacklevel=2 ) return _read_table_internal(source, columns=columns,