This is an automated email from the ASF dual-hosted git repository.

rok pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/arrow.git


The following commit(s) were added to refs/heads/main by this push:
     new dc906048914 GH-51220: [Python][Parquet] Accept write_table options in 
dataset writer (#51461)
dc906048914 is described below

commit dc906048914bce2b7d98d1c66cc97c8c720d439e
Author: tadeja <[email protected]>
AuthorDate: Tue Sep 29 20:41:28 2026 +0200

    GH-51220: [Python][Parquet] Accept write_table options in dataset writer 
(#51461)
    
    ### Rationale for this change
    Fix #51220 - Parquet `write_table` options are not all accepted in dataset 
`ParquetFileWriteOptions` so `pyarrow.parquet.write_to_dataset( ...)` raises 
`TypeError: unexpected parquet write option:`
    for `bloom_filter_options` ,  `write_time_adjusted_to_utc` or `store_schema`
    
    (Similar to past #37469)
    ### What changes are included in this PR?
    Add `bloom_filter_options`, `store_schema`, `write_time_adjusted_to_utc` to 
`ParquetFileWriteOptions`,
    also explicitly expose `bloom_filter_options` in 
`pyarrow.parquet.ParquetWriter`.
    Test writer options in `make_write_options`, and test three newly supported 
options in `write_to_dataset`
    
    ### Are these changes tested?
    Yes. CI also passes.
    
    ### Are there any user-facing changes?
    Yes, users may now use `bloom_filter_options` ,  
`write_time_adjusted_to_utc` and `store_schema`
    in `pyarrow.parquet.write_to_dataset(...)` and in 
`ParquetFileFormat.make_write_options(...)` without raising TypeError.
    
    ### Was AI used for this PR?
    
    **PR code and description written by:**
    
    - [x] Human
    - [ ] AI
    
    **Reviewed before submission by:**
    
    - [x] Human
    - [x] AI
    - [ ] Not reviewed
    * GitHub Issue: #51220
    
    Authored-by: Tadeja Kadunc <[email protected]>
    Signed-off-by: Rok Mihevc <[email protected]>
---
 python/pyarrow/_dataset_parquet.pyx          | 10 +++++++++-
 python/pyarrow/parquet/core.py               |  2 ++
 python/pyarrow/tests/parquet/test_dataset.py | 27 +++++++++++++++++++++++++++
 3 files changed, 38 insertions(+), 1 deletion(-)

diff --git a/python/pyarrow/_dataset_parquet.pyx 
b/python/pyarrow/_dataset_parquet.pyx
index fac8e96df85..43eca8957be 100644
--- a/python/pyarrow/_dataset_parquet.pyx
+++ b/python/pyarrow/_dataset_parquet.pyx
@@ -620,6 +620,8 @@ cdef class ParquetFileWriteOptions(FileWriteOptions):
             "coerce_timestamps",
             "allow_truncated_timestamps",
             "use_compliant_nested_type",
+            "store_schema",
+            "write_time_adjusted_to_utc",
         }
 
         setters = set()
@@ -661,6 +663,7 @@ cdef class ParquetFileWriteOptions(FileWriteOptions):
             sorting_columns=self._properties["sorting_columns"],
             
store_decimal_as_integer=self._properties["store_decimal_as_integer"],
             
use_content_defined_chunking=self._properties["use_content_defined_chunking"],
+            bloom_filter_options=self._properties["bloom_filter_options"],
         )
 
     def _set_arrow_properties(self):
@@ -677,7 +680,9 @@ cdef class ParquetFileWriteOptions(FileWriteOptions):
             writer_engine_version="V2",
             use_compliant_nested_type=(
                 self._properties["use_compliant_nested_type"]
-            )
+            ),
+            store_schema=self._properties["store_schema"],
+            
write_time_adjusted_to_utc=self._properties["write_time_adjusted_to_utc"],
         )
 
     def _set_encryption_config(self):
@@ -706,6 +711,7 @@ cdef class ParquetFileWriteOptions(FileWriteOptions):
             coerce_timestamps=None,
             allow_truncated_timestamps=False,
             use_compliant_nested_type=True,
+            store_schema=True,
             encryption_properties=None,
             write_batch_size=None,
             dictionary_pagesize_limit=None,
@@ -715,6 +721,8 @@ cdef class ParquetFileWriteOptions(FileWriteOptions):
             sorting_columns=None,
             store_decimal_as_integer=False,
             use_content_defined_chunking=False,
+            write_time_adjusted_to_utc=False,
+            bloom_filter_options=None,
         )
 
         self._set_properties()
diff --git a/python/pyarrow/parquet/core.py b/python/pyarrow/parquet/core.py
index 6f837d21861..4de88fe2299 100644
--- a/python/pyarrow/parquet/core.py
+++ b/python/pyarrow/parquet/core.py
@@ -1088,6 +1088,7 @@ Examples
                  store_decimal_as_integer=False,
                  write_time_adjusted_to_utc=False,
                  max_rows_per_page=None,
+                 bloom_filter_options=None,
                  use_content_defined_chunking=False,
                  **options):
         if use_deprecated_int96_timestamps is None:
@@ -1144,6 +1145,7 @@ Examples
             store_decimal_as_integer=store_decimal_as_integer,
             write_time_adjusted_to_utc=write_time_adjusted_to_utc,
             max_rows_per_page=max_rows_per_page,
+            bloom_filter_options=bloom_filter_options,
             use_content_defined_chunking=use_content_defined_chunking,
             **options)
         self.is_open = True
diff --git a/python/pyarrow/tests/parquet/test_dataset.py 
b/python/pyarrow/tests/parquet/test_dataset.py
index 88603d9b5b8..28b63663796 100644
--- a/python/pyarrow/tests/parquet/test_dataset.py
+++ b/python/pyarrow/tests/parquet/test_dataset.py
@@ -1312,6 +1312,33 @@ def 
test_parquet_write_to_dataset_exposed_keywords(tempdir):
     assert paths_written_set == expected_paths
 
 
+def test_write_table_options_in_make_write_options():
+    import pyarrow.dataset as ds
+
+    # Test write_table options are in sync with the dataset writer.
+    # Any option not in ParquetFileWriteOptions raises TypeError in 
make_write_options
+    not_writer_args = {"table", "where", "row_group_size", "filesystem", 
"flavor"}
+    options = {
+        name: parameter.default
+        for name, parameter in 
inspect.signature(pq.write_table).parameters.items()
+        if name not in not_writer_args
+        and parameter.kind != inspect.Parameter.VAR_KEYWORD
+    }
+    ds.ParquetFileFormat().make_write_options(**options)
+
+
+def test_write_to_dataset_options(tempdir):
+    table = pa.table({"a": [1, 2, 3],
+                      "t": pa.array([1, 2, 3], pa.time32("ms"))})
+    pq.write_to_dataset(table, tempdir, store_schema=False,
+                        write_time_adjusted_to_utc=True,
+                        bloom_filter_options={"a": True})
+    metadata = pq.read_metadata(next(tempdir.glob("*.parquet")))
+    assert b'ARROW:schema' not in (metadata.metadata or {})
+    assert 'isAdjustedToUTC=true' in 
str(metadata.schema.column(1).logical_type)
+    assert metadata.row_group(0).column(0).bloom_filter_offset is not None
+
+
 @pytest.mark.parametrize("write_dataset_kwarg", (
     ("create_dir", True),
     ("create_dir", False),

Reply via email to