This is an automated email from the ASF dual-hosted git repository.
rok pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/arrow.git
The following commit(s) were added to refs/heads/main by this push:
new dc906048914 GH-51220: [Python][Parquet] Accept write_table options in
dataset writer (#51461)
dc906048914 is described below
commit dc906048914bce2b7d98d1c66cc97c8c720d439e
Author: tadeja <[email protected]>
AuthorDate: Tue Sep 29 20:41:28 2026 +0200
GH-51220: [Python][Parquet] Accept write_table options in dataset writer
(#51461)
### Rationale for this change
Fix #51220 - Parquet `write_table` options are not all accepted in dataset
`ParquetFileWriteOptions` so `pyarrow.parquet.write_to_dataset( ...)` raises
`TypeError: unexpected parquet write option:`
for `bloom_filter_options` , `write_time_adjusted_to_utc` or `store_schema`
(Similar to past #37469)
### What changes are included in this PR?
Add `bloom_filter_options`, `store_schema`, `write_time_adjusted_to_utc` to
`ParquetFileWriteOptions`,
also explicitly expose `bloom_filter_options` in
`pyarrow.parquet.ParquetWriter`.
Test writer options in `make_write_options`, and test three newly supported
options in `write_to_dataset`
### Are these changes tested?
Yes. CI also passes.
### Are there any user-facing changes?
Yes, users may now use `bloom_filter_options` ,
`write_time_adjusted_to_utc` and `store_schema`
in `pyarrow.parquet.write_to_dataset(...)` and in
`ParquetFileFormat.make_write_options(...)` without raising TypeError.
### Was AI used for this PR?
**PR code and description written by:**
- [x] Human
- [ ] AI
**Reviewed before submission by:**
- [x] Human
- [x] AI
- [ ] Not reviewed
* GitHub Issue: #51220
Authored-by: Tadeja Kadunc <[email protected]>
Signed-off-by: Rok Mihevc <[email protected]>
---
python/pyarrow/_dataset_parquet.pyx | 10 +++++++++-
python/pyarrow/parquet/core.py | 2 ++
python/pyarrow/tests/parquet/test_dataset.py | 27 +++++++++++++++++++++++++++
3 files changed, 38 insertions(+), 1 deletion(-)
diff --git a/python/pyarrow/_dataset_parquet.pyx
b/python/pyarrow/_dataset_parquet.pyx
index fac8e96df85..43eca8957be 100644
--- a/python/pyarrow/_dataset_parquet.pyx
+++ b/python/pyarrow/_dataset_parquet.pyx
@@ -620,6 +620,8 @@ cdef class ParquetFileWriteOptions(FileWriteOptions):
"coerce_timestamps",
"allow_truncated_timestamps",
"use_compliant_nested_type",
+ "store_schema",
+ "write_time_adjusted_to_utc",
}
setters = set()
@@ -661,6 +663,7 @@ cdef class ParquetFileWriteOptions(FileWriteOptions):
sorting_columns=self._properties["sorting_columns"],
store_decimal_as_integer=self._properties["store_decimal_as_integer"],
use_content_defined_chunking=self._properties["use_content_defined_chunking"],
+ bloom_filter_options=self._properties["bloom_filter_options"],
)
def _set_arrow_properties(self):
@@ -677,7 +680,9 @@ cdef class ParquetFileWriteOptions(FileWriteOptions):
writer_engine_version="V2",
use_compliant_nested_type=(
self._properties["use_compliant_nested_type"]
- )
+ ),
+ store_schema=self._properties["store_schema"],
+
write_time_adjusted_to_utc=self._properties["write_time_adjusted_to_utc"],
)
def _set_encryption_config(self):
@@ -706,6 +711,7 @@ cdef class ParquetFileWriteOptions(FileWriteOptions):
coerce_timestamps=None,
allow_truncated_timestamps=False,
use_compliant_nested_type=True,
+ store_schema=True,
encryption_properties=None,
write_batch_size=None,
dictionary_pagesize_limit=None,
@@ -715,6 +721,8 @@ cdef class ParquetFileWriteOptions(FileWriteOptions):
sorting_columns=None,
store_decimal_as_integer=False,
use_content_defined_chunking=False,
+ write_time_adjusted_to_utc=False,
+ bloom_filter_options=None,
)
self._set_properties()
diff --git a/python/pyarrow/parquet/core.py b/python/pyarrow/parquet/core.py
index 6f837d21861..4de88fe2299 100644
--- a/python/pyarrow/parquet/core.py
+++ b/python/pyarrow/parquet/core.py
@@ -1088,6 +1088,7 @@ Examples
store_decimal_as_integer=False,
write_time_adjusted_to_utc=False,
max_rows_per_page=None,
+ bloom_filter_options=None,
use_content_defined_chunking=False,
**options):
if use_deprecated_int96_timestamps is None:
@@ -1144,6 +1145,7 @@ Examples
store_decimal_as_integer=store_decimal_as_integer,
write_time_adjusted_to_utc=write_time_adjusted_to_utc,
max_rows_per_page=max_rows_per_page,
+ bloom_filter_options=bloom_filter_options,
use_content_defined_chunking=use_content_defined_chunking,
**options)
self.is_open = True
diff --git a/python/pyarrow/tests/parquet/test_dataset.py
b/python/pyarrow/tests/parquet/test_dataset.py
index 88603d9b5b8..28b63663796 100644
--- a/python/pyarrow/tests/parquet/test_dataset.py
+++ b/python/pyarrow/tests/parquet/test_dataset.py
@@ -1312,6 +1312,33 @@ def
test_parquet_write_to_dataset_exposed_keywords(tempdir):
assert paths_written_set == expected_paths
+def test_write_table_options_in_make_write_options():
+ import pyarrow.dataset as ds
+
+ # Test write_table options are in sync with the dataset writer.
+ # Any option not in ParquetFileWriteOptions raises TypeError in
make_write_options
+ not_writer_args = {"table", "where", "row_group_size", "filesystem",
"flavor"}
+ options = {
+ name: parameter.default
+ for name, parameter in
inspect.signature(pq.write_table).parameters.items()
+ if name not in not_writer_args
+ and parameter.kind != inspect.Parameter.VAR_KEYWORD
+ }
+ ds.ParquetFileFormat().make_write_options(**options)
+
+
+def test_write_to_dataset_options(tempdir):
+ table = pa.table({"a": [1, 2, 3],
+ "t": pa.array([1, 2, 3], pa.time32("ms"))})
+ pq.write_to_dataset(table, tempdir, store_schema=False,
+ write_time_adjusted_to_utc=True,
+ bloom_filter_options={"a": True})
+ metadata = pq.read_metadata(next(tempdir.glob("*.parquet")))
+ assert b'ARROW:schema' not in (metadata.metadata or {})
+ assert 'isAdjustedToUTC=true' in
str(metadata.schema.column(1).logical_type)
+ assert metadata.row_group(0).column(0).bloom_filter_offset is not None
+
+
@pytest.mark.parametrize("write_dataset_kwarg", (
("create_dir", True),
("create_dir", False),