This is an automated email from the ASF dual-hosted git repository.

etseidl pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/arrow-rs.git


The following commit(s) were added to refs/heads/main by this push:
     new b9b1d5005a feat(parquet): provide direct access to dictionary pages 
(#10420)
b9b1d5005a is described below

commit b9b1d5005ac6b4b1b828d652674ed5db48275498
Author: Oleg V. Kozlyuk <[email protected]>
AuthorDate: Fri Sep 25 21:18:18 2026 +0200

    feat(parquet): provide direct access to dictionary pages (#10420)
    
    # Which issue does this PR close?
    
    - Closes #9010.
    - Supersedes #9011 (closed as stale)
    
    # Rationale for this change
    
    This change provides low-level API necessary for enabling
    dictionary-based pruning in DataFusion:
    https://github.com/apache/datafusion/pull/23851
    
    # What changes are included in this PR?
    
    - `parquet::file::metadata::dictionary::decode_dictionary_page`: decodes
    a BYTE_ARRAY dictionary page (Thrift header parse, decompress, PLAIN
    decode) into a `Utf8`/`Binary` Arrow array.
    - `ParquetMetaDataReader::read_column_dictionary` (sync) and
    `read_column_dictionary_async` (async) to fetch and decode a given row
    group/column's dictionary page from a `ParquetMetaData`, returning
    `Ok(None)` if the chunk has no dictionary page.
    - `ParquetRecordBatchStreamBuilder::get_row_group_column_dictionary`
    convenience method mirroring `get_row_group_column_bloom_filter`.
    
    # Are these changes tested?
    
    Yes: a round-trip unit test for a dictionary-encoded string column, a
    non-BYTE_ARRAY rejection test, and sync + async reader tests that decode
    a real dictionary page written through `ArrowWriter`.
    
    # Are there any user-facing changes?
    
    Yes, three new public APIs (see above). No changes to existing API.
    
    ---------
    
    Co-authored-by: Claude Sonnet 5 <[email protected]>
    Co-authored-by: Ed Seidl <[email protected]>
---
 parquet/src/arrow/array_reader/mod.rs   |   3 +
 parquet/src/arrow/async_reader/mod.rs   | 191 +++++++++++-
 parquet/src/arrow/mod.rs                |   4 +
 parquet/src/file/metadata/dictionary.rs | 498 ++++++++++++++++++++++++++++++++
 parquet/src/file/metadata/mod.rs        |   3 +
 parquet/src/file/metadata/reader.rs     |  86 ++++++
 parquet/src/file/serialized_reader.rs   |  52 ++--
 7 files changed, 814 insertions(+), 23 deletions(-)

diff --git a/parquet/src/arrow/array_reader/mod.rs 
b/parquet/src/arrow/array_reader/mod.rs
index 32fb90d2e1..2d9b982379 100644
--- a/parquet/src/arrow/array_reader/mod.rs
+++ b/parquet/src/arrow/array_reader/mod.rs
@@ -55,6 +55,9 @@ pub(crate) mod test_util;
 use crate::file::metadata::RowGroupMetaData;
 pub use builder::{ArrayReaderBuilder, CacheOptions, CacheOptionsBuilder};
 pub use byte_array::make_byte_array_reader;
+// Re-exported (beyond the `experimental` feature) so 
`file::metadata::dictionary`
+// can PLAIN-decode a raw dictionary page without duplicating this logic.
+pub(crate) use byte_array::ByteArrayDecoderPlain;
 pub use byte_array_dictionary::make_byte_array_dictionary_reader;
 #[cfg_attr(not(feature = "experimental"), expect(unused_imports))]
 pub use byte_view_array::make_byte_view_array_reader;
diff --git a/parquet/src/arrow/async_reader/mod.rs 
b/parquet/src/arrow/async_reader/mod.rs
index 24978e7f4c..7d5a07f369 100644
--- a/parquet/src/arrow/async_reader/mod.rs
+++ b/parquet/src/arrow/async_reader/mod.rs
@@ -33,7 +33,7 @@ use futures::future::{BoxFuture, FutureExt};
 use futures::stream::Stream;
 use tokio::io::{AsyncRead, AsyncReadExt, AsyncSeek, AsyncSeekExt};
 
-use arrow_array::RecordBatch;
+use arrow_array::{ArrayRef, RecordBatch};
 use arrow_schema::{Schema, SchemaRef};
 
 use crate::arrow::arrow_reader::{
@@ -675,6 +675,39 @@ impl<T: AsyncFileReader + Send + 'static> 
ParquetRecordBatchStreamBuilder<T> {
         self
     }
 
+    /// Read and decode the dictionary page for a column in a row group, if 
any.
+    ///
+    /// Returns `Ok(None)` if the column chunk has no dictionary page, or if
+    /// its physical type is not `BYTE_ARRAY` (the only physical type
+    /// currently supported).
+    ///
+    /// The returned array contains raw `Binary` values, even for columns
+    /// annotated as strings. Callers can compare byte slices directly or
+    /// convert values to UTF-8 explicitly.
+    ///
+    /// This can be used to inspect dictionary values when selecting or pruning
+    /// row groups before passing the selected indices to
+    /// [`ParquetRecordBatchStreamBuilder::with_row_groups`].
+    ///
+    /// Note this does not verify that the *entire* column chunk is
+    /// dictionary-encoded -- callers that need that guarantee (e.g. to treat
+    /// the dictionary as an exhaustive set of the column's values) should
+    /// check
+    /// 
[`crate::file::metadata::ColumnChunkMetaData::page_encoding_stats_mask`].
+    pub async fn get_column_chunk_dictionary(
+        &mut self,
+        row_group_idx: usize,
+        column_idx: usize,
+    ) -> Result<Option<ArrayRef>> {
+        ParquetMetaDataReader::read_column_dictionary_async(
+            &mut self.input.0,
+            &self.metadata,
+            row_group_idx,
+            column_idx,
+        )
+        .await
+    }
+
     /// Build a new [`ParquetRecordBatchStream`]
     ///
     /// See examples on [`ParquetRecordBatchStreamBuilder::new`]
@@ -972,6 +1005,7 @@ mod tests {
     use crate::arrow::arrow_reader::{ArrowReaderMetadata, ArrowReaderOptions};
     use crate::arrow::schema::virtual_type::RowNumber;
     use crate::arrow::{ArrowWriter, AsyncArrowWriter, ProjectionMask};
+    use crate::basic::Encoding;
     use crate::file::metadata::PageIndexPolicy;
     use crate::file::metadata::ParquetMetaDataReader;
     use crate::file::metadata::page_index::PageIndex;
@@ -982,8 +1016,8 @@ mod tests {
     use arrow_array::cast::AsArray;
     use arrow_array::types::Int32Type;
     use arrow_array::{
-        Array, ArrayRef, BooleanArray, Int32Array, RecordBatchReader, Scalar, 
StringArray,
-        StructArray, UInt64Array,
+        Array, ArrayRef, BinaryArray, BooleanArray, Int32Array, 
RecordBatchReader, Scalar,
+        StringArray, StructArray, UInt64Array,
     };
     use arrow_schema::{DataType, Field, Schema};
     use futures::{StreamExt, TryStreamExt};
@@ -1126,6 +1160,157 @@ mod tests {
         );
     }
 
+    #[tokio::test]
+    async fn test_get_column_chunk_dictionary() {
+        let schema = Arc::new(Schema::new(vec![Field::new("s", DataType::Utf8, 
false)]));
+        let values: Vec<&str> = ["alpha", "beta", "gamma"]
+            .iter()
+            .copied()
+            .cycle()
+            .take(30)
+            .collect();
+        let array: ArrayRef = Arc::new(StringArray::from(values));
+        let batch = RecordBatch::try_new(schema.clone(), vec![array]).unwrap();
+
+        let props = WriterProperties::builder()
+            .set_dictionary_enabled(true)
+            .build();
+        let mut buf = Vec::new();
+        {
+            let mut writer = ArrowWriter::try_new(&mut buf, schema, 
Some(props)).unwrap();
+            writer.write(&batch).unwrap();
+            writer.close().unwrap();
+        }
+        let data = Bytes::from(buf);
+
+        let direct_metadata = ParquetMetaDataReader::new()
+            .parse_and_finish(&data)
+            .unwrap();
+        let mut direct_reader = TestReader::new(data.clone());
+        let direct = ParquetMetaDataReader::read_column_dictionary_async(
+            &mut direct_reader,
+            &direct_metadata,
+            0,
+            0,
+        )
+        .await
+        .unwrap()
+        .unwrap();
+        let direct = direct.as_any().downcast_ref::<BinaryArray>().unwrap();
+        assert_eq!(direct.value(0), b"alpha");
+
+        let async_reader = TestReader::new(data);
+        let mut builder = ParquetRecordBatchStreamBuilder::new(async_reader)
+            .await
+            .unwrap();
+
+        let dictionary = builder
+            .get_column_chunk_dictionary(0, 0)
+            .await
+            .unwrap()
+            .unwrap();
+        let dictionary = 
dictionary.as_any().downcast_ref::<BinaryArray>().unwrap();
+        let dictionary_values: Vec<&[u8]> = dictionary.iter().map(|v| 
v.unwrap()).collect();
+        assert_eq!(
+            dictionary_values,
+            vec![b"alpha".as_slice(), b"beta", b"gamma"]
+        );
+    }
+
+    // This test demonstrates row group pruning using dictionary pages,
+    // verifying that data pages of skipped row groups are not read
+    #[tokio::test]
+    async fn test_dictionary_selects_row_groups_without_reading_skipped_data() 
{
+        // Write two row groups, each dictionary-encoded and containing a 
single
+        // distinct string repeated 30 times: row group 0 is all "skip", row
+        // group 1 is all "target".
+        let schema = Arc::new(Schema::new(vec![Field::new("s", DataType::Utf8, 
false)]));
+        let row_group_values = ["skip", "target"];
+        let props = WriterProperties::builder()
+            .set_dictionary_enabled(true)
+            .build();
+        let mut buf = Vec::new();
+        {
+            let mut writer = ArrowWriter::try_new(&mut buf, schema.clone(), 
Some(props)).unwrap();
+            for value in row_group_values {
+                let array: ArrayRef = Arc::new(StringArray::from(vec![value; 
30]));
+                let batch = RecordBatch::try_new(schema.clone(), 
vec![array]).unwrap();
+                writer.write(&batch).unwrap();
+                writer.flush().unwrap();
+            }
+            writer.close().unwrap();
+        }
+
+        let async_reader = TestReader::new(Bytes::from(buf));
+        // `requests` records the byte ranges fetched from the underlying 
reader,
+        // which we later use to check that we read only the needed page
+        let requests = async_reader.requests.clone();
+        let mut builder = ParquetRecordBatchStreamBuilder::new(async_reader)
+            .await
+            .unwrap();
+        let metadata = builder.metadata().clone();
+        assert_eq!(metadata.num_row_groups(), 2);
+
+        // For each row group, fetch and decode just the dictionary page
+        // to decide whether to read the whole row group
+        let mut selected_row_groups = Vec::new();
+        let mut dictionary_ranges = Vec::new();
+        for row_group_idx in 0..metadata.num_row_groups() {
+            let column = metadata.row_group(row_group_idx).column(0);
+            let encoding_mask = column.page_encoding_stats_mask().unwrap();
+            assert!(
+                encoding_mask.is_only(Encoding::PLAIN_DICTIONARY)
+                    || encoding_mask.is_only(Encoding::RLE_DICTIONARY)
+            );
+
+            let dictionary_start = column.dictionary_page_offset().unwrap() as 
usize;
+            let data_start = column.data_page_offset() as usize;
+            dictionary_ranges.push(dictionary_start..data_start);
+
+            let dictionary = builder
+                .get_column_chunk_dictionary(row_group_idx, 0)
+                .await
+                .unwrap()
+                .unwrap();
+            let dictionary = dictionary.as_binary::<i32>();
+            if dictionary
+                .iter()
+                .any(|value| value == Some(b"target".as_slice()))
+            {
+                selected_row_groups.push(row_group_idx);
+            }
+        }
+        assert_eq!(selected_row_groups, vec![1]);
+
+        // Read the filtered row group and verify the values
+        let batches: Vec<_> = builder
+            .with_row_groups(selected_row_groups)
+            .build()
+            .unwrap()
+            .try_collect()
+            .await
+            .unwrap();
+        let values: Vec<_> = batches
+            .iter()
+            .flat_map(|batch| batch.column(0).as_string::<i32>().iter())
+            .collect();
+        assert_eq!(values, vec![Some("target"); 30]);
+
+        // Finally, verify none of the requests overlapped the data pages
+        // of the skipped row group
+        let skipped_column = metadata.row_group(0).column(0);
+        let (skipped_start, skipped_len) = skipped_column.byte_range();
+        let skipped_data_range =
+            skipped_column.data_page_offset() as usize..(skipped_start + 
skipped_len) as usize;
+        let requests = requests.lock().unwrap();
+        for dictionary_range in dictionary_ranges {
+            assert!(requests.contains(&dictionary_range));
+        }
+        assert!(requests.iter().all(|request| {
+            request.end <= skipped_data_range.start || request.start >= 
skipped_data_range.end
+        }));
+    }
+
     #[tokio::test]
     async fn test_async_reader_with_next_row_group() {
         let testdata = arrow::util::test_util::parquet_test_data();
diff --git a/parquet/src/arrow/mod.rs b/parquet/src/arrow/mod.rs
index e5cbafea5e..d25126cbd7 100644
--- a/parquet/src/arrow/mod.rs
+++ b/parquet/src/arrow/mod.rs
@@ -186,9 +186,13 @@
 pub mod array_reader;
 #[cfg(not(feature = "experimental"))]
 mod array_reader;
+// Re-exported (beyond the `experimental` feature) so 
`file::metadata::dictionary`
+// can PLAIN-decode a raw dictionary page without duplicating this logic.
+pub(crate) use array_reader::ByteArrayDecoderPlain;
 pub mod arrow_reader;
 pub mod arrow_writer;
 mod buffer;
+pub(crate) use buffer::offset_buffer::OffsetBuffer;
 mod decoder;
 
 #[cfg(feature = "async")]
diff --git a/parquet/src/file/metadata/dictionary.rs 
b/parquet/src/file/metadata/dictionary.rs
new file mode 100644
index 0000000000..809fbff355
--- /dev/null
+++ b/parquet/src/file/metadata/dictionary.rs
@@ -0,0 +1,498 @@
+// Licensed to the Apache Software Foundation (ASF) under one
+// or more contributor license agreements.  See the NOTICE file
+// distributed with this work for additional information
+// regarding copyright ownership.  The ASF licenses this file
+// to you under the Apache License, Version 2.0 (the
+// "License"); you may not use this file except in compliance
+// with the License.  You may obtain a copy of the License at
+//
+//   http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing,
+// software distributed under the License is distributed on an
+// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+// KIND, either express or implied.  See the License for the
+// specific language governing permissions and limitations
+// under the License.
+
+//! Decoding a column chunk's dictionary page directly into an Arrow array,
+//! independent of the row-by-row [`ArrayReader`] machinery.
+//!
+//! This is useful for callers that want the *set* of distinct values stored
+//! in a dictionary-encoded column chunk without reading any data pages, e.g.
+//! to prune a row group when the query predicate's literals are known not to
+//! be in the dictionary.
+//!
+//! [`ArrayReader`]: crate::arrow::array_reader::ArrayReader
+
+use crate::arrow::{ByteArrayDecoderPlain, OffsetBuffer};
+use crate::basic::{Encoding, PageType, Type as PhysicalType};
+use crate::column::page::Page;
+use crate::compression::{CodecOptions, create_codec};
+#[cfg(feature = "encryption")]
+use crate::encryption::decrypt::CryptoContext;
+use crate::errors::{ParquetError, Result};
+#[cfg(feature = "encryption")]
+use crate::file::metadata::ColumnChunkMetaData;
+use crate::file::metadata::ParquetMetaData;
+use crate::file::serialized_reader::{
+    SerializedPageReaderContext, decode_page, read_page_header_len_from_bytes, 
verify_page_size,
+};
+use arrow_array::ArrayRef;
+use arrow_schema::DataType as ArrowType;
+use bytes::Bytes;
+#[cfg(feature = "encryption")]
+use std::sync::Arc;
+
+/// Decodes the dictionary page of a column chunk into an [`ArrayRef`].
+///
+/// `buffer` must contain the entire dictionary page, byte-for-byte, i.e. the
+/// range `[dictionary_page_offset, data_page_offset)` of the column chunk.
+///
+/// Only `BYTE_ARRAY` columns are currently supported; other physical types
+/// return an error. The returned array never contains nulls: dictionary
+/// pages only store the distinct non-null values, with nulls represented via
+/// definition levels in the data pages.
+/// The returned array has `Binary` values, regardless of the column's logical 
type.
+///
+/// Note this only decodes whatever dictionary page is present -- it does
+/// **not** verify that the entire column chunk is dictionary-encoded (i.e.
+/// that every value in the chunk is drawn from this dictionary). Callers
+/// that need that guarantee (for example, to use the dictionary as an exact
+/// membership index) must check that themselves, e.g. via
+/// [`crate::file::metadata::ColumnChunkMetaData::page_encoding_stats_mask`].
+pub(crate) fn decode_dictionary_page(
+    buffer: Bytes,
+    parquet_meta_data: &ParquetMetaData,
+    row_group_idx: usize,
+    column_idx: usize,
+) -> Result<ArrayRef> {
+    let column_metadata = parquet_meta_data
+        .row_group(row_group_idx)
+        .column(column_idx);
+    let column_descriptor = column_metadata.column_descr();
+
+    if column_descriptor.physical_type() != PhysicalType::BYTE_ARRAY {
+        return Err(ParquetError::General(format!(
+            "decode_dictionary_page only supports BYTE_ARRAY columns, got {}",
+            column_descriptor.physical_type()
+        )));
+    }
+
+    // Dictionary pages are subject to the same modular encryption as data
+    // pages: both the page header and the page body may be ciphertext, so
+    // we must route through the same crypto-aware header/data path that
+    // `SerializedPageReader` uses rather than parsing the header directly.
+    let page_context = SerializedPageReaderContext {
+        read_stats: true,
+        #[cfg(feature = "encryption")]
+        crypto_context: dictionary_page_crypto_context(
+            parquet_meta_data,
+            column_metadata,
+            row_group_idx,
+            column_idx,
+        )?,
+    };
+
+    let (consumed, header) =
+        read_page_header_len_from_bytes(&page_context, buffer.as_ref(), 0, 
true)?;
+    if header.r#type != PageType::DICTIONARY_PAGE {
+        return Err(ParquetError::General(format!(
+            "Expected a dictionary page, found {:?}",
+            header.r#type
+        )));
+    }
+
+    // `compressed_page_size` comes from the (possibly maliciously crafted)
+    // file header; `verify_page_size` bounds-checks it against what we
+    // actually fetched before we slice, instead of trusting it blindly.
+    let remaining = (buffer.len() - consumed) as u64;
+    verify_page_size(
+        header.compressed_page_size,
+        header.uncompressed_page_size,
+        remaining,
+    )?;
+    let compressed_size = header.compressed_page_size as usize;
+    let page_buf = buffer.slice(consumed..consumed + compressed_size);
+    let page_buf = page_context.decrypt_page_data(page_buf, 0, true)?;
+
+    let mut decompressor = create_codec(column_metadata.compression(), 
&CodecOptions::default())?;
+    let page = decode_page(
+        header,
+        page_buf,
+        column_descriptor.physical_type(),
+        decompressor.as_mut(),
+    )?;
+    let Page::DictionaryPage {
+        buf,
+        num_values,
+        encoding,
+        ..
+    } = page
+    else {
+        return Err(ParquetError::General(
+            "Expected a dictionary page".to_string(),
+        ));
+    };
+    let num_values = num_values as usize;
+
+    // The dictionary page is always PLAIN-encoded, regardless of what the
+    // data pages' encoding is (RLE_DICTIONARY/PLAIN_DICTIONARY only describe
+    // how *data* pages reference the dictionary by index).
+    if encoding != Encoding::PLAIN {
+        return Err(ParquetError::General(format!(
+            "Dictionary page encoding must be PLAIN, got {encoding:?}"
+        )));
+    }
+    let mut decoder = ByteArrayDecoderPlain::new(buf, num_values, 
Some(num_values), false);
+    let mut offsets = OffsetBuffer::<i32>::with_capacity(num_values);
+    decoder.read(&mut offsets, usize::MAX)?;
+    if offsets.len() != num_values {
+        return Err(ParquetError::General(format!(
+            "Expected {num_values} dictionary values, decoded {}",
+            offsets.len()
+        )));
+    }
+
+    Ok(offsets.into_array(None, ArrowType::Binary))
+}
+
+/// Builds the crypto context needed to decrypt the dictionary page of
+/// `column_metadata`, or `None` if the file (or this column) isn't encrypted.
+#[cfg(feature = "encryption")]
+fn dictionary_page_crypto_context(
+    parquet_meta_data: &ParquetMetaData,
+    column_metadata: &ColumnChunkMetaData,
+    row_group_idx: usize,
+    column_idx: usize,
+) -> Result<Option<Arc<CryptoContext>>> {
+    let Some(file_decryptor) = parquet_meta_data.file_decryptor() else {
+        return Ok(None);
+    };
+    let Some(crypto_metadata) = column_metadata.crypto_metadata() else {
+        return Ok(None);
+    };
+    let ordinal = parquet_meta_data
+        .row_group(row_group_idx)
+        .ordinal()
+        .ok_or_else(|| {
+            ParquetError::General("Encrypted row group is missing its file 
ordinal".to_string())
+        })?;
+    let ordinal = usize::try_from(ordinal).map_err(|_| {
+        ParquetError::General("Encrypted row group has an invalid file 
ordinal".to_string())
+    })?;
+    let crypto_context =
+        CryptoContext::for_column(file_decryptor, crypto_metadata, ordinal, 
column_idx)?
+            .for_dictionary_page();
+    Ok(Some(Arc::new(crypto_context)))
+}
+
+#[cfg(test)]
+mod tests {
+    use super::*;
+    use crate::arrow::ArrowWriter;
+    use crate::basic::Encoding;
+    use crate::file::metadata::ParquetMetaDataReader;
+    use crate::file::properties::WriterProperties;
+    use crate::file::reader::{ChunkReader, FileReader, SerializedFileReader};
+    use crate::parquet_thrift::{ThriftCompactOutputProtocol, WriteThrift};
+    use arrow_array::{Array, BinaryArray, RecordBatch, StringArray};
+    use arrow_schema::{Field, Schema};
+    use std::sync::Arc;
+
+    fn write_dictionary_encoded_strings(values: &[&str]) -> Bytes {
+        let schema = Arc::new(Schema::new(vec![Field::new("s", 
ArrowType::Utf8, false)]));
+        let array = 
Arc::new(StringArray::from_iter_values(values.iter().copied()));
+        let batch = RecordBatch::try_new(schema.clone(), vec![array]).unwrap();
+
+        let props = WriterProperties::builder()
+            .set_dictionary_enabled(true)
+            .build();
+        let mut buf = Vec::new();
+        {
+            let mut writer = ArrowWriter::try_new(&mut buf, schema, 
Some(props)).unwrap();
+            writer.write(&batch).unwrap();
+            writer.close().unwrap();
+        }
+        Bytes::from(buf)
+    }
+
+    #[test]
+    fn decode_dictionary_page_round_trips_strings() {
+        let distinct_values = ["alpha", "beta", "gamma"];
+        // Repeat so the column is worth dictionary-encoding but the
+        // dictionary itself only contains the distinct values.
+        let values: Vec<&str> = 
distinct_values.iter().copied().cycle().take(30).collect();
+        let data = write_dictionary_encoded_strings(&values);
+
+        let reader = SerializedFileReader::new(data.clone()).unwrap();
+        let metadata = reader.metadata();
+        let column_metadata = metadata.row_group(0).column(0);
+
+        assert!(
+            column_metadata.dictionary_page_offset().is_some(),
+            "expected the column chunk to be dictionary-encoded"
+        );
+
+        let start = column_metadata.dictionary_page_offset().unwrap() as u64;
+        let end = column_metadata.data_page_offset() as u64;
+        let buffer = data.get_bytes(start, (end - start) as usize).unwrap();
+
+        let array = decode_dictionary_page(buffer, metadata, 0, 0).unwrap();
+        let array = array.as_any().downcast_ref::<BinaryArray>().unwrap();
+        let decoded: Vec<&[u8]> = array.iter().map(|v| v.unwrap()).collect();
+        assert_eq!(decoded, distinct_values.map(str::as_bytes));
+    }
+
+    #[test]
+    fn decode_dictionary_page_errors_on_truncated_buffer() {
+        let distinct_values = ["alpha", "beta", "gamma"];
+        let values: Vec<&str> = 
distinct_values.iter().copied().cycle().take(30).collect();
+        let data = write_dictionary_encoded_strings(&values);
+
+        let reader = SerializedFileReader::new(data.clone()).unwrap();
+        let metadata = reader.metadata();
+        let column_metadata = metadata.row_group(0).column(0);
+
+        let start = column_metadata.dictionary_page_offset().unwrap() as u64;
+        let end = column_metadata.data_page_offset() as u64;
+        let buffer = data.get_bytes(start, (end - start) as usize).unwrap();
+
+        // Simulate a truncated/malformed file: the page header's declared
+        // `compressed_page_size` no longer fits in what was actually
+        // fetched. This must return an error rather than panic while
+        // slicing (`Bytes::slice` panics on out-of-bounds ranges).
+        let truncated = buffer.slice(..buffer.len() - 1);
+        let err = decode_dictionary_page(truncated, metadata, 0, 
0).unwrap_err();
+        assert!(
+            matches!(err, ParquetError::EOF(_)),
+            "unexpected error: {err}"
+        );
+    }
+
+    fn dictionary_page_with_header_change(
+        data: &Bytes,
+        change: impl FnOnce(&mut crate::file::metadata::thrift::PageHeader),
+    ) -> (Bytes, ParquetMetaData) {
+        let reader = SerializedFileReader::new(data.clone()).unwrap();
+        let metadata = reader.metadata().clone();
+        let column = metadata.row_group(0).column(0);
+        let start = column.dictionary_page_offset().unwrap() as u64;
+        let end = column.data_page_offset() as u64;
+        let buffer = data.get_bytes(start, (end - start) as usize).unwrap();
+        let context = SerializedPageReaderContext {
+            read_stats: true,
+            #[cfg(feature = "encryption")]
+            crypto_context: None,
+        };
+        let (header_len, mut header) =
+            read_page_header_len_from_bytes(&context, &buffer, 0, 
true).unwrap();
+        change(&mut header);
+        let mut changed = Vec::new();
+        header
+            .write_thrift(&mut ThriftCompactOutputProtocol::new(&mut changed))
+            .unwrap();
+        changed.extend_from_slice(&buffer[header_len..]);
+        (Bytes::from(changed), metadata)
+    }
+
+    #[test]
+    fn decode_dictionary_page_rejects_missing_values() {
+        let data = write_dictionary_encoded_strings(&["alpha", "beta", 
"alpha"]);
+        let (buffer, metadata) = dictionary_page_with_header_change(&data, 
|header| {
+            header.dictionary_page_header.as_mut().unwrap().num_values += 1;
+        });
+        let err = decode_dictionary_page(buffer, &metadata, 0, 0).unwrap_err();
+        assert!(err.to_string().contains("dictionary values"), "{err}");
+    }
+
+    #[test]
+    fn decode_dictionary_page_rejects_non_plain_encoding() {
+        let data = write_dictionary_encoded_strings(&["alpha", "beta", 
"alpha"]);
+        let (buffer, metadata) = dictionary_page_with_header_change(&data, 
|header| {
+            header.dictionary_page_header.as_mut().unwrap().encoding = 
Encoding::RLE_DICTIONARY;
+        });
+        let err = decode_dictionary_page(buffer, &metadata, 0, 0).unwrap_err();
+        assert!(err.to_string().contains("PLAIN"), "{err}");
+    }
+
+    #[test]
+    fn decode_dictionary_page_rejects_non_byte_array() {
+        let schema = Arc::new(Schema::new(vec![Field::new("i", 
ArrowType::Int32, false)]));
+        let array = Arc::new(arrow_array::Int32Array::from(vec![1, 2, 3]));
+        let batch = RecordBatch::try_new(schema.clone(), vec![array]).unwrap();
+        let mut buf = Vec::new();
+        {
+            let mut writer = ArrowWriter::try_new(&mut buf, schema, 
None).unwrap();
+            writer.write(&batch).unwrap();
+            writer.close().unwrap();
+        }
+        let data = Bytes::from(buf);
+
+        let reader = SerializedFileReader::new(data).unwrap();
+        let metadata = reader.metadata();
+
+        let err = decode_dictionary_page(Bytes::new(), metadata, 0, 
0).unwrap_err();
+        assert!(err.to_string().contains("BYTE_ARRAY"));
+    }
+
+    #[test]
+    fn read_column_dictionary_round_trips_via_metadata_reader() {
+        let distinct_values = ["alpha", "beta", "gamma"];
+        let values: Vec<&str> = 
distinct_values.iter().copied().cycle().take(30).collect();
+        let data = write_dictionary_encoded_strings(&values);
+
+        let reader = SerializedFileReader::new(data.clone()).unwrap();
+        let metadata = reader.metadata();
+
+        let array = ParquetMetaDataReader::read_column_dictionary(&data, 
metadata, 0, 0)
+            .unwrap()
+            .unwrap();
+        let array = array.as_any().downcast_ref::<BinaryArray>().unwrap();
+        let decoded: Vec<&[u8]> = array.iter().map(|v| v.unwrap()).collect();
+        assert_eq!(decoded, distinct_values.map(str::as_bytes));
+    }
+
+    #[cfg(feature = "encryption")]
+    #[test]
+    fn read_column_dictionary_round_trips_with_encryption() {
+        use crate::encryption::decrypt::FileDecryptionProperties;
+        use crate::encryption::encrypt::FileEncryptionProperties;
+        const FOOTER_KEY: &[u8] = b"0123456789012345";
+
+        let distinct_values = ["alpha", "beta", "gamma"];
+        let values: Vec<&str> = 
distinct_values.iter().copied().cycle().take(30).collect();
+
+        let schema = Arc::new(Schema::new(vec![Field::new("s", 
ArrowType::Utf8, false)]));
+        let array = 
Arc::new(StringArray::from_iter_values(values.iter().copied()));
+        let batch = RecordBatch::try_new(schema.clone(), vec![array]).unwrap();
+
+        let encryption_properties = 
FileEncryptionProperties::builder(FOOTER_KEY.to_vec())
+            .build()
+            .unwrap();
+        let props = WriterProperties::builder()
+            .set_dictionary_enabled(true)
+            .with_file_encryption_properties(encryption_properties)
+            .build();
+        let mut buf = Vec::new();
+        {
+            let mut writer = ArrowWriter::try_new(&mut buf, schema, 
Some(props)).unwrap();
+            writer.write(&batch).unwrap();
+            writer.close().unwrap();
+        }
+        let data = Bytes::from(buf);
+
+        let decryption_properties = 
FileDecryptionProperties::builder(FOOTER_KEY.to_vec())
+            .build()
+            .unwrap();
+        let metadata = ParquetMetaDataReader::new()
+            .with_decryption_properties(Some(decryption_properties))
+            .parse_and_finish(&data)
+            .unwrap();
+
+        let array = ParquetMetaDataReader::read_column_dictionary(&data, 
&metadata, 0, 0)
+            .unwrap()
+            .unwrap();
+        let array = array.as_any().downcast_ref::<BinaryArray>().unwrap();
+        let decoded: Vec<&[u8]> = array.iter().map(|v| v.unwrap()).collect();
+        assert_eq!(decoded, distinct_values.map(str::as_bytes));
+    }
+
+    #[cfg(feature = "encryption")]
+    #[test]
+    fn read_column_dictionary_uses_file_ordinal_after_filtering() {
+        use crate::encryption::decrypt::FileDecryptionProperties;
+        use crate::encryption::encrypt::FileEncryptionProperties;
+        use crate::file::metadata::ParquetMetaDataBuilder;
+
+        const FOOTER_KEY: &[u8] = b"0123456789012345";
+        let schema = Arc::new(Schema::new(vec![Field::new("s", 
ArrowType::Utf8, false)]));
+        let props = WriterProperties::builder()
+            .set_dictionary_enabled(true)
+            .with_file_encryption_properties(
+                FileEncryptionProperties::builder(FOOTER_KEY.to_vec())
+                    .build()
+                    .unwrap(),
+            )
+            .build();
+        let mut buf = Vec::new();
+        {
+            let mut writer = ArrowWriter::try_new(&mut buf, schema.clone(), 
Some(props)).unwrap();
+            for value in ["first", "second"] {
+                let array = 
Arc::new(StringArray::from_iter_values(std::iter::repeat_n(
+                    value, 30,
+                )));
+                let batch = RecordBatch::try_new(schema.clone(), 
vec![array]).unwrap();
+                writer.write(&batch).unwrap();
+                writer.flush().unwrap();
+            }
+            writer.close().unwrap();
+        }
+        let data = Bytes::from(buf);
+        let metadata = ParquetMetaDataReader::new()
+            .with_decryption_properties(Some(
+                FileDecryptionProperties::builder(FOOTER_KEY.to_vec())
+                    .build()
+                    .unwrap(),
+            ))
+            .parse_and_finish(&data)
+            .unwrap();
+        assert_eq!(metadata.row_group(1).ordinal(), Some(1));
+        let second = metadata.row_group(1).clone();
+        let filtered = ParquetMetaDataBuilder::new_from_metadata(metadata)
+            .set_row_groups(vec![second])
+            .build();
+        let array = ParquetMetaDataReader::read_column_dictionary(&data, 
&filtered, 0, 0)
+            .unwrap()
+            .unwrap();
+        let array = array.as_any().downcast_ref::<BinaryArray>().unwrap();
+        assert_eq!(array.value(0), b"second");
+
+        let invalid = filtered
+            .row_group(0)
+            .clone()
+            .into_builder()
+            .set_ordinal(-1)
+            .build()
+            .unwrap();
+        let invalid_metadata = 
ParquetMetaDataBuilder::new_from_metadata(filtered)
+            .set_row_groups(vec![invalid])
+            .build();
+        let err = ParquetMetaDataReader::read_column_dictionary(&data, 
&invalid_metadata, 0, 0)
+            .unwrap_err();
+        assert!(err.to_string().contains("invalid file ordinal"), "{err}");
+    }
+
+    #[test]
+    fn read_column_dictionary_returns_none_without_dictionary_page() {
+        use crate::file::metadata::ParquetMetaDataReader;
+
+        let schema = Arc::new(Schema::new(vec![Field::new("s", 
ArrowType::Utf8, false)]));
+        let array = Arc::new(StringArray::from_iter_values(["a", "b", "c"]));
+        let batch = RecordBatch::try_new(schema.clone(), vec![array]).unwrap();
+
+        let props = WriterProperties::builder()
+            .set_dictionary_enabled(false)
+            .build();
+        let mut buf = Vec::new();
+        {
+            let mut writer = ArrowWriter::try_new(&mut buf, schema, 
Some(props)).unwrap();
+            writer.write(&batch).unwrap();
+            writer.close().unwrap();
+        }
+        let data = Bytes::from(buf);
+
+        let reader = SerializedFileReader::new(data.clone()).unwrap();
+        let metadata = reader.metadata();
+        assert!(
+            metadata
+                .row_group(0)
+                .column(0)
+                .dictionary_page_offset()
+                .is_none()
+        );
+
+        let result = ParquetMetaDataReader::read_column_dictionary(&data, 
metadata, 0, 0).unwrap();
+        assert!(result.is_none());
+    }
+}
diff --git a/parquet/src/file/metadata/mod.rs b/parquet/src/file/metadata/mod.rs
index 166a7eb7ab..a12fbdd1de 100644
--- a/parquet/src/file/metadata/mod.rs
+++ b/parquet/src/file/metadata/mod.rs
@@ -49,6 +49,9 @@
 //! Please see [`external_metadata.rs`]
 //!
 //! [`external_metadata.rs`]: 
https://github.com/apache/arrow-rs/tree/master/parquet/examples/external_metadata.rs
+//!
+#[cfg(feature = "arrow")]
+mod dictionary;
 mod footer_tail;
 mod memory;
 mod options;
diff --git a/parquet/src/file/metadata/reader.rs 
b/parquet/src/file/metadata/reader.rs
index 3ffce28523..24bc5c982b 100644
--- a/parquet/src/file/metadata/reader.rs
+++ b/parquet/src/file/metadata/reader.rs
@@ -19,6 +19,8 @@
 use crate::encryption::decrypt::FileDecryptionProperties;
 use crate::errors::{ParquetError, Result};
 use crate::file::FOOTER_SIZE;
+#[cfg(feature = "arrow")]
+use crate::file::metadata::dictionary::decode_dictionary_page;
 use crate::file::metadata::parser::decode_metadata;
 use crate::file::metadata::thrift::parquet_schema_from_bytes;
 use crate::file::metadata::{
@@ -26,6 +28,8 @@ use crate::file::metadata::{
 };
 use crate::file::reader::ChunkReader;
 use crate::schema::types::SchemaDescriptor;
+#[cfg(feature = "arrow")]
+use arrow_array::ArrayRef;
 use bytes::Bytes;
 use std::sync::Arc;
 use std::{io::Read, ops::Range};
@@ -472,6 +476,53 @@ impl ParquetMetaDataReader {
         self.load_page_index_with_remainder(fetch, None).await
     }
 
+    /// Reads and decodes the dictionary page of a column chunk into an Arrow 
array.
+    ///
+    /// Returns `Ok(None)` if the column chunk has no dictionary page, or if
+    /// its physical type is not `BYTE_ARRAY` (the only physical type
+    /// currently supported).
+    ///
+    /// The returned array contains raw `Binary` values, even for columns
+    /// annotated as strings. Callers can compare byte slices directly or
+    /// convert values to UTF-8 explicitly.
+    ///
+    /// This can be used to inspect dictionary values when selecting or pruning
+    /// row groups before reading their data pages.
+    ///
+    /// Note this does not verify that the *entire* column chunk is
+    /// dictionary-encoded (i.e. that the dictionary contains every value in
+    /// the chunk) -- callers that need that guarantee should check
+    /// 
[`crate::file::metadata::ColumnChunkMetaData::page_encoding_stats_mask`].
+    #[cfg(feature = "arrow")]
+    pub fn read_column_dictionary<R: ChunkReader>(
+        reader: &R,
+        metadata: &ParquetMetaData,
+        row_group_idx: usize,
+        column_idx: usize,
+    ) -> Result<Option<ArrayRef>> {
+        let Some(range) = dictionary_page_byte_range(metadata, row_group_idx, 
column_idx)? else {
+            return Ok(None);
+        };
+        let length = usize::try_from(range.end - range.start)?;
+        let buffer = reader.get_bytes(range.start, length)?;
+        decode_dictionary_page(buffer, metadata, row_group_idx, 
column_idx).map(Some)
+    }
+
+    /// Asynchronous version of [`Self::read_column_dictionary`].
+    #[cfg(all(feature = "async", feature = "arrow"))]
+    pub async fn read_column_dictionary_async<F: MetadataFetch>(
+        mut fetch: F,
+        metadata: &ParquetMetaData,
+        row_group_idx: usize,
+        column_idx: usize,
+    ) -> Result<Option<ArrayRef>> {
+        let Some(range) = dictionary_page_byte_range(metadata, row_group_idx, 
column_idx)? else {
+            return Ok(None);
+        };
+        let buffer = fetch.fetch(range).await?;
+        decode_dictionary_page(buffer, metadata, row_group_idx, 
column_idx).map(Some)
+    }
+
     #[cfg(all(feature = "async", feature = "arrow"))]
     async fn load_page_index_with_remainder<F: MetadataFetch>(
         &mut self,
@@ -841,6 +892,41 @@ fn parse_index_data(push_decoder: &mut 
ParquetMetaDataPushDecoder) -> Result<Par
     }
 }
 
+/// Returns the `[start, end)` byte range of a column chunk's dictionary
+/// page, or `None` if the column chunk has no dictionary page or is not a
+/// `BYTE_ARRAY` column (the only physical type [`decode_dictionary_page`]
+/// currently supports).
+#[cfg(feature = "arrow")]
+fn dictionary_page_byte_range(
+    metadata: &ParquetMetaData,
+    row_group_idx: usize,
+    column_idx: usize,
+) -> Result<Option<Range<u64>>> {
+    let column_metadata = metadata.row_group(row_group_idx).column(column_idx);
+    let column_descriptor = column_metadata.column_descr();
+
+    if column_descriptor.physical_type() != crate::basic::Type::BYTE_ARRAY {
+        return Ok(None);
+    }
+    let Some(start) = column_metadata.dictionary_page_offset() else {
+        return Ok(None);
+    };
+    let start: u64 = start
+        .try_into()
+        .map_err(|_| ParquetError::General("Dictionary page offset is 
invalid".to_string()))?;
+    let end: u64 = column_metadata
+        .data_page_offset()
+        .try_into()
+        .map_err(|_| ParquetError::General("Data page offset is 
invalid".to_string()))?;
+    if end < start {
+        return Err(ParquetError::General(
+            "Data page offset precedes dictionary page offset".to_string(),
+        ));
+    }
+
+    Ok(Some(start..end))
+}
+
 #[cfg(test)]
 mod tests {
     use super::*;
diff --git a/parquet/src/file/serialized_reader.rs 
b/parquet/src/file/serialized_reader.rs
index 9f411b2dc9..7345fb9645 100644
--- a/parquet/src/file/serialized_reader.rs
+++ b/parquet/src/file/serialized_reader.rs
@@ -563,12 +563,12 @@ enum SerializedPageReaderState {
 }
 
 #[derive(Default)]
-struct SerializedPageReaderContext {
+pub(crate) struct SerializedPageReaderContext {
     /// Controls decoding of page-level statistics
-    read_stats: bool,
+    pub(crate) read_stats: bool,
     /// Crypto context carrying objects required for decryption
     #[cfg(feature = "encryption")]
-    crypto_context: Option<Arc<CryptoContext>>,
+    pub(crate) crypto_context: Option<Arc<CryptoContext>>,
 }
 
 /// A serialized implementation for Parquet [`PageReader`].
@@ -781,23 +781,30 @@ impl<R: ChunkReader> SerializedPageReader<R> {
         let header = context.read_page_header(&mut tracked, page_index, 
dictionary_page)?;
         Ok((tracked.bytes_read, header))
     }
+}
 
-    fn read_page_header_len_from_bytes(
-        context: &SerializedPageReaderContext,
-        buffer: &[u8],
-        page_index: usize,
-        dictionary_page: bool,
-    ) -> Result<(usize, PageHeader)> {
-        let mut input = std::io::Cursor::new(buffer);
-        let header = context.read_page_header(&mut input, page_index, 
dictionary_page)?;
-        let header_len = input.position() as usize;
-        Ok((header_len, header))
-    }
+/// Reads (and decrypts, if `context` carries a crypto context) the page 
header stored
+/// at the front of `buffer`, returning the header and the number of bytes of 
`buffer`
+/// it occupies.
+///
+/// This is exposed for callers that need to decode a single page directly 
from an
+/// already-fetched byte range, outside of the normal [`SerializedPageReader`] 
iteration
+/// -- e.g. decoding a dictionary page standalone.
+pub(crate) fn read_page_header_len_from_bytes(
+    context: &SerializedPageReaderContext,
+    buffer: &[u8],
+    page_index: usize,
+    dictionary_page: bool,
+) -> Result<(usize, PageHeader)> {
+    let mut input = std::io::Cursor::new(buffer);
+    let header = context.read_page_header(&mut input, page_index, 
dictionary_page)?;
+    let header_len = input.position() as usize;
+    Ok((header_len, header))
 }
 
 #[cfg(not(feature = "encryption"))]
 impl SerializedPageReaderContext {
-    fn read_page_header<T: Read>(
+    pub(crate) fn read_page_header<T: Read>(
         &self,
         input: &mut T,
         _page_index: usize,
@@ -811,7 +818,7 @@ impl SerializedPageReaderContext {
         }
     }
 
-    fn decrypt_page_data<T>(
+    pub(crate) fn decrypt_page_data<T>(
         &self,
         buffer: T,
         _page_index: usize,
@@ -823,7 +830,7 @@ impl SerializedPageReaderContext {
 
 #[cfg(feature = "encryption")]
 impl SerializedPageReaderContext {
-    fn read_page_header<T: Read>(
+    pub(crate) fn read_page_header<T: Read>(
         &self,
         input: &mut T,
         page_index: usize,
@@ -861,7 +868,12 @@ impl SerializedPageReaderContext {
         }
     }
 
-    fn decrypt_page_data<T>(&self, buffer: T, page_index: usize, 
dictionary_page: bool) -> Result<T>
+    pub(crate) fn decrypt_page_data<T>(
+        &self,
+        buffer: T,
+        page_index: usize,
+        dictionary_page: bool,
+    ) -> Result<T>
     where
         T: AsRef<[u8]>,
         T: From<Vec<u8>>,
@@ -907,7 +919,7 @@ fn verify_page_header_len(header_len: usize, 
remaining_bytes: u64) -> Result<()>
     Ok(())
 }
 
-fn verify_page_size(
+pub(crate) fn verify_page_size(
     compressed_size: i32,
     uncompressed_size: i32,
     remaining_bytes: u64,
@@ -1001,7 +1013,7 @@ impl<R: ChunkReader> PageReader for 
SerializedPageReader<R> {
                     let page_len = 
usize::try_from(front.compressed_page_size)?;
                     let buffer = self.reader.get_bytes(front.offset as u64, 
page_len)?;
 
-                    let (offset, header) = 
Self::read_page_header_len_from_bytes(
+                    let (offset, header) = read_page_header_len_from_bytes(
                         &self.context,
                         buffer.as_ref(),
                         *page_index,

Reply via email to