This is an automated email from the ASF dual-hosted git repository.
etseidl pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/arrow-rs.git
The following commit(s) were added to refs/heads/main by this push:
new b9b1d5005a feat(parquet): provide direct access to dictionary pages
(#10420)
b9b1d5005a is described below
commit b9b1d5005ac6b4b1b828d652674ed5db48275498
Author: Oleg V. Kozlyuk <[email protected]>
AuthorDate: Fri Sep 25 21:18:18 2026 +0200
feat(parquet): provide direct access to dictionary pages (#10420)
# Which issue does this PR close?
- Closes #9010.
- Supersedes #9011 (closed as stale)
# Rationale for this change
This change provides low-level API necessary for enabling
dictionary-based pruning in DataFusion:
https://github.com/apache/datafusion/pull/23851
# What changes are included in this PR?
- `parquet::file::metadata::dictionary::decode_dictionary_page`: decodes
a BYTE_ARRAY dictionary page (Thrift header parse, decompress, PLAIN
decode) into a `Utf8`/`Binary` Arrow array.
- `ParquetMetaDataReader::read_column_dictionary` (sync) and
`read_column_dictionary_async` (async) to fetch and decode a given row
group/column's dictionary page from a `ParquetMetaData`, returning
`Ok(None)` if the chunk has no dictionary page.
- `ParquetRecordBatchStreamBuilder::get_row_group_column_dictionary`
convenience method mirroring `get_row_group_column_bloom_filter`.
# Are these changes tested?
Yes: a round-trip unit test for a dictionary-encoded string column, a
non-BYTE_ARRAY rejection test, and sync + async reader tests that decode
a real dictionary page written through `ArrowWriter`.
# Are there any user-facing changes?
Yes, three new public APIs (see above). No changes to existing API.
---------
Co-authored-by: Claude Sonnet 5 <[email protected]>
Co-authored-by: Ed Seidl <[email protected]>
---
parquet/src/arrow/array_reader/mod.rs | 3 +
parquet/src/arrow/async_reader/mod.rs | 191 +++++++++++-
parquet/src/arrow/mod.rs | 4 +
parquet/src/file/metadata/dictionary.rs | 498 ++++++++++++++++++++++++++++++++
parquet/src/file/metadata/mod.rs | 3 +
parquet/src/file/metadata/reader.rs | 86 ++++++
parquet/src/file/serialized_reader.rs | 52 ++--
7 files changed, 814 insertions(+), 23 deletions(-)
diff --git a/parquet/src/arrow/array_reader/mod.rs
b/parquet/src/arrow/array_reader/mod.rs
index 32fb90d2e1..2d9b982379 100644
--- a/parquet/src/arrow/array_reader/mod.rs
+++ b/parquet/src/arrow/array_reader/mod.rs
@@ -55,6 +55,9 @@ pub(crate) mod test_util;
use crate::file::metadata::RowGroupMetaData;
pub use builder::{ArrayReaderBuilder, CacheOptions, CacheOptionsBuilder};
pub use byte_array::make_byte_array_reader;
+// Re-exported (beyond the `experimental` feature) so
`file::metadata::dictionary`
+// can PLAIN-decode a raw dictionary page without duplicating this logic.
+pub(crate) use byte_array::ByteArrayDecoderPlain;
pub use byte_array_dictionary::make_byte_array_dictionary_reader;
#[cfg_attr(not(feature = "experimental"), expect(unused_imports))]
pub use byte_view_array::make_byte_view_array_reader;
diff --git a/parquet/src/arrow/async_reader/mod.rs
b/parquet/src/arrow/async_reader/mod.rs
index 24978e7f4c..7d5a07f369 100644
--- a/parquet/src/arrow/async_reader/mod.rs
+++ b/parquet/src/arrow/async_reader/mod.rs
@@ -33,7 +33,7 @@ use futures::future::{BoxFuture, FutureExt};
use futures::stream::Stream;
use tokio::io::{AsyncRead, AsyncReadExt, AsyncSeek, AsyncSeekExt};
-use arrow_array::RecordBatch;
+use arrow_array::{ArrayRef, RecordBatch};
use arrow_schema::{Schema, SchemaRef};
use crate::arrow::arrow_reader::{
@@ -675,6 +675,39 @@ impl<T: AsyncFileReader + Send + 'static>
ParquetRecordBatchStreamBuilder<T> {
self
}
+ /// Read and decode the dictionary page for a column in a row group, if
any.
+ ///
+ /// Returns `Ok(None)` if the column chunk has no dictionary page, or if
+ /// its physical type is not `BYTE_ARRAY` (the only physical type
+ /// currently supported).
+ ///
+ /// The returned array contains raw `Binary` values, even for columns
+ /// annotated as strings. Callers can compare byte slices directly or
+ /// convert values to UTF-8 explicitly.
+ ///
+ /// This can be used to inspect dictionary values when selecting or pruning
+ /// row groups before passing the selected indices to
+ /// [`ParquetRecordBatchStreamBuilder::with_row_groups`].
+ ///
+ /// Note this does not verify that the *entire* column chunk is
+ /// dictionary-encoded -- callers that need that guarantee (e.g. to treat
+ /// the dictionary as an exhaustive set of the column's values) should
+ /// check
+ ///
[`crate::file::metadata::ColumnChunkMetaData::page_encoding_stats_mask`].
+ pub async fn get_column_chunk_dictionary(
+ &mut self,
+ row_group_idx: usize,
+ column_idx: usize,
+ ) -> Result<Option<ArrayRef>> {
+ ParquetMetaDataReader::read_column_dictionary_async(
+ &mut self.input.0,
+ &self.metadata,
+ row_group_idx,
+ column_idx,
+ )
+ .await
+ }
+
/// Build a new [`ParquetRecordBatchStream`]
///
/// See examples on [`ParquetRecordBatchStreamBuilder::new`]
@@ -972,6 +1005,7 @@ mod tests {
use crate::arrow::arrow_reader::{ArrowReaderMetadata, ArrowReaderOptions};
use crate::arrow::schema::virtual_type::RowNumber;
use crate::arrow::{ArrowWriter, AsyncArrowWriter, ProjectionMask};
+ use crate::basic::Encoding;
use crate::file::metadata::PageIndexPolicy;
use crate::file::metadata::ParquetMetaDataReader;
use crate::file::metadata::page_index::PageIndex;
@@ -982,8 +1016,8 @@ mod tests {
use arrow_array::cast::AsArray;
use arrow_array::types::Int32Type;
use arrow_array::{
- Array, ArrayRef, BooleanArray, Int32Array, RecordBatchReader, Scalar,
StringArray,
- StructArray, UInt64Array,
+ Array, ArrayRef, BinaryArray, BooleanArray, Int32Array,
RecordBatchReader, Scalar,
+ StringArray, StructArray, UInt64Array,
};
use arrow_schema::{DataType, Field, Schema};
use futures::{StreamExt, TryStreamExt};
@@ -1126,6 +1160,157 @@ mod tests {
);
}
+ #[tokio::test]
+ async fn test_get_column_chunk_dictionary() {
+ let schema = Arc::new(Schema::new(vec![Field::new("s", DataType::Utf8,
false)]));
+ let values: Vec<&str> = ["alpha", "beta", "gamma"]
+ .iter()
+ .copied()
+ .cycle()
+ .take(30)
+ .collect();
+ let array: ArrayRef = Arc::new(StringArray::from(values));
+ let batch = RecordBatch::try_new(schema.clone(), vec![array]).unwrap();
+
+ let props = WriterProperties::builder()
+ .set_dictionary_enabled(true)
+ .build();
+ let mut buf = Vec::new();
+ {
+ let mut writer = ArrowWriter::try_new(&mut buf, schema,
Some(props)).unwrap();
+ writer.write(&batch).unwrap();
+ writer.close().unwrap();
+ }
+ let data = Bytes::from(buf);
+
+ let direct_metadata = ParquetMetaDataReader::new()
+ .parse_and_finish(&data)
+ .unwrap();
+ let mut direct_reader = TestReader::new(data.clone());
+ let direct = ParquetMetaDataReader::read_column_dictionary_async(
+ &mut direct_reader,
+ &direct_metadata,
+ 0,
+ 0,
+ )
+ .await
+ .unwrap()
+ .unwrap();
+ let direct = direct.as_any().downcast_ref::<BinaryArray>().unwrap();
+ assert_eq!(direct.value(0), b"alpha");
+
+ let async_reader = TestReader::new(data);
+ let mut builder = ParquetRecordBatchStreamBuilder::new(async_reader)
+ .await
+ .unwrap();
+
+ let dictionary = builder
+ .get_column_chunk_dictionary(0, 0)
+ .await
+ .unwrap()
+ .unwrap();
+ let dictionary =
dictionary.as_any().downcast_ref::<BinaryArray>().unwrap();
+ let dictionary_values: Vec<&[u8]> = dictionary.iter().map(|v|
v.unwrap()).collect();
+ assert_eq!(
+ dictionary_values,
+ vec![b"alpha".as_slice(), b"beta", b"gamma"]
+ );
+ }
+
+ // This test demonstrates row group pruning using dictionary pages,
+ // verifying that data pages of skipped row groups are not read
+ #[tokio::test]
+ async fn test_dictionary_selects_row_groups_without_reading_skipped_data()
{
+ // Write two row groups, each dictionary-encoded and containing a
single
+ // distinct string repeated 30 times: row group 0 is all "skip", row
+ // group 1 is all "target".
+ let schema = Arc::new(Schema::new(vec![Field::new("s", DataType::Utf8,
false)]));
+ let row_group_values = ["skip", "target"];
+ let props = WriterProperties::builder()
+ .set_dictionary_enabled(true)
+ .build();
+ let mut buf = Vec::new();
+ {
+ let mut writer = ArrowWriter::try_new(&mut buf, schema.clone(),
Some(props)).unwrap();
+ for value in row_group_values {
+ let array: ArrayRef = Arc::new(StringArray::from(vec![value;
30]));
+ let batch = RecordBatch::try_new(schema.clone(),
vec![array]).unwrap();
+ writer.write(&batch).unwrap();
+ writer.flush().unwrap();
+ }
+ writer.close().unwrap();
+ }
+
+ let async_reader = TestReader::new(Bytes::from(buf));
+ // `requests` records the byte ranges fetched from the underlying
reader,
+ // which we later use to check that we read only the needed page
+ let requests = async_reader.requests.clone();
+ let mut builder = ParquetRecordBatchStreamBuilder::new(async_reader)
+ .await
+ .unwrap();
+ let metadata = builder.metadata().clone();
+ assert_eq!(metadata.num_row_groups(), 2);
+
+ // For each row group, fetch and decode just the dictionary page
+ // to decide whether to read the whole row group
+ let mut selected_row_groups = Vec::new();
+ let mut dictionary_ranges = Vec::new();
+ for row_group_idx in 0..metadata.num_row_groups() {
+ let column = metadata.row_group(row_group_idx).column(0);
+ let encoding_mask = column.page_encoding_stats_mask().unwrap();
+ assert!(
+ encoding_mask.is_only(Encoding::PLAIN_DICTIONARY)
+ || encoding_mask.is_only(Encoding::RLE_DICTIONARY)
+ );
+
+ let dictionary_start = column.dictionary_page_offset().unwrap() as
usize;
+ let data_start = column.data_page_offset() as usize;
+ dictionary_ranges.push(dictionary_start..data_start);
+
+ let dictionary = builder
+ .get_column_chunk_dictionary(row_group_idx, 0)
+ .await
+ .unwrap()
+ .unwrap();
+ let dictionary = dictionary.as_binary::<i32>();
+ if dictionary
+ .iter()
+ .any(|value| value == Some(b"target".as_slice()))
+ {
+ selected_row_groups.push(row_group_idx);
+ }
+ }
+ assert_eq!(selected_row_groups, vec![1]);
+
+ // Read the filtered row group and verify the values
+ let batches: Vec<_> = builder
+ .with_row_groups(selected_row_groups)
+ .build()
+ .unwrap()
+ .try_collect()
+ .await
+ .unwrap();
+ let values: Vec<_> = batches
+ .iter()
+ .flat_map(|batch| batch.column(0).as_string::<i32>().iter())
+ .collect();
+ assert_eq!(values, vec![Some("target"); 30]);
+
+ // Finally, verify none of the requests overlapped the data pages
+ // of the skipped row group
+ let skipped_column = metadata.row_group(0).column(0);
+ let (skipped_start, skipped_len) = skipped_column.byte_range();
+ let skipped_data_range =
+ skipped_column.data_page_offset() as usize..(skipped_start +
skipped_len) as usize;
+ let requests = requests.lock().unwrap();
+ for dictionary_range in dictionary_ranges {
+ assert!(requests.contains(&dictionary_range));
+ }
+ assert!(requests.iter().all(|request| {
+ request.end <= skipped_data_range.start || request.start >=
skipped_data_range.end
+ }));
+ }
+
#[tokio::test]
async fn test_async_reader_with_next_row_group() {
let testdata = arrow::util::test_util::parquet_test_data();
diff --git a/parquet/src/arrow/mod.rs b/parquet/src/arrow/mod.rs
index e5cbafea5e..d25126cbd7 100644
--- a/parquet/src/arrow/mod.rs
+++ b/parquet/src/arrow/mod.rs
@@ -186,9 +186,13 @@
pub mod array_reader;
#[cfg(not(feature = "experimental"))]
mod array_reader;
+// Re-exported (beyond the `experimental` feature) so
`file::metadata::dictionary`
+// can PLAIN-decode a raw dictionary page without duplicating this logic.
+pub(crate) use array_reader::ByteArrayDecoderPlain;
pub mod arrow_reader;
pub mod arrow_writer;
mod buffer;
+pub(crate) use buffer::offset_buffer::OffsetBuffer;
mod decoder;
#[cfg(feature = "async")]
diff --git a/parquet/src/file/metadata/dictionary.rs
b/parquet/src/file/metadata/dictionary.rs
new file mode 100644
index 0000000000..809fbff355
--- /dev/null
+++ b/parquet/src/file/metadata/dictionary.rs
@@ -0,0 +1,498 @@
+// Licensed to the Apache Software Foundation (ASF) under one
+// or more contributor license agreements. See the NOTICE file
+// distributed with this work for additional information
+// regarding copyright ownership. The ASF licenses this file
+// to you under the Apache License, Version 2.0 (the
+// "License"); you may not use this file except in compliance
+// with the License. You may obtain a copy of the License at
+//
+// http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing,
+// software distributed under the License is distributed on an
+// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+// KIND, either express or implied. See the License for the
+// specific language governing permissions and limitations
+// under the License.
+
+//! Decoding a column chunk's dictionary page directly into an Arrow array,
+//! independent of the row-by-row [`ArrayReader`] machinery.
+//!
+//! This is useful for callers that want the *set* of distinct values stored
+//! in a dictionary-encoded column chunk without reading any data pages, e.g.
+//! to prune a row group when the query predicate's literals are known not to
+//! be in the dictionary.
+//!
+//! [`ArrayReader`]: crate::arrow::array_reader::ArrayReader
+
+use crate::arrow::{ByteArrayDecoderPlain, OffsetBuffer};
+use crate::basic::{Encoding, PageType, Type as PhysicalType};
+use crate::column::page::Page;
+use crate::compression::{CodecOptions, create_codec};
+#[cfg(feature = "encryption")]
+use crate::encryption::decrypt::CryptoContext;
+use crate::errors::{ParquetError, Result};
+#[cfg(feature = "encryption")]
+use crate::file::metadata::ColumnChunkMetaData;
+use crate::file::metadata::ParquetMetaData;
+use crate::file::serialized_reader::{
+ SerializedPageReaderContext, decode_page, read_page_header_len_from_bytes,
verify_page_size,
+};
+use arrow_array::ArrayRef;
+use arrow_schema::DataType as ArrowType;
+use bytes::Bytes;
+#[cfg(feature = "encryption")]
+use std::sync::Arc;
+
+/// Decodes the dictionary page of a column chunk into an [`ArrayRef`].
+///
+/// `buffer` must contain the entire dictionary page, byte-for-byte, i.e. the
+/// range `[dictionary_page_offset, data_page_offset)` of the column chunk.
+///
+/// Only `BYTE_ARRAY` columns are currently supported; other physical types
+/// return an error. The returned array never contains nulls: dictionary
+/// pages only store the distinct non-null values, with nulls represented via
+/// definition levels in the data pages.
+/// The returned array has `Binary` values, regardless of the column's logical
type.
+///
+/// Note this only decodes whatever dictionary page is present -- it does
+/// **not** verify that the entire column chunk is dictionary-encoded (i.e.
+/// that every value in the chunk is drawn from this dictionary). Callers
+/// that need that guarantee (for example, to use the dictionary as an exact
+/// membership index) must check that themselves, e.g. via
+/// [`crate::file::metadata::ColumnChunkMetaData::page_encoding_stats_mask`].
+pub(crate) fn decode_dictionary_page(
+ buffer: Bytes,
+ parquet_meta_data: &ParquetMetaData,
+ row_group_idx: usize,
+ column_idx: usize,
+) -> Result<ArrayRef> {
+ let column_metadata = parquet_meta_data
+ .row_group(row_group_idx)
+ .column(column_idx);
+ let column_descriptor = column_metadata.column_descr();
+
+ if column_descriptor.physical_type() != PhysicalType::BYTE_ARRAY {
+ return Err(ParquetError::General(format!(
+ "decode_dictionary_page only supports BYTE_ARRAY columns, got {}",
+ column_descriptor.physical_type()
+ )));
+ }
+
+ // Dictionary pages are subject to the same modular encryption as data
+ // pages: both the page header and the page body may be ciphertext, so
+ // we must route through the same crypto-aware header/data path that
+ // `SerializedPageReader` uses rather than parsing the header directly.
+ let page_context = SerializedPageReaderContext {
+ read_stats: true,
+ #[cfg(feature = "encryption")]
+ crypto_context: dictionary_page_crypto_context(
+ parquet_meta_data,
+ column_metadata,
+ row_group_idx,
+ column_idx,
+ )?,
+ };
+
+ let (consumed, header) =
+ read_page_header_len_from_bytes(&page_context, buffer.as_ref(), 0,
true)?;
+ if header.r#type != PageType::DICTIONARY_PAGE {
+ return Err(ParquetError::General(format!(
+ "Expected a dictionary page, found {:?}",
+ header.r#type
+ )));
+ }
+
+ // `compressed_page_size` comes from the (possibly maliciously crafted)
+ // file header; `verify_page_size` bounds-checks it against what we
+ // actually fetched before we slice, instead of trusting it blindly.
+ let remaining = (buffer.len() - consumed) as u64;
+ verify_page_size(
+ header.compressed_page_size,
+ header.uncompressed_page_size,
+ remaining,
+ )?;
+ let compressed_size = header.compressed_page_size as usize;
+ let page_buf = buffer.slice(consumed..consumed + compressed_size);
+ let page_buf = page_context.decrypt_page_data(page_buf, 0, true)?;
+
+ let mut decompressor = create_codec(column_metadata.compression(),
&CodecOptions::default())?;
+ let page = decode_page(
+ header,
+ page_buf,
+ column_descriptor.physical_type(),
+ decompressor.as_mut(),
+ )?;
+ let Page::DictionaryPage {
+ buf,
+ num_values,
+ encoding,
+ ..
+ } = page
+ else {
+ return Err(ParquetError::General(
+ "Expected a dictionary page".to_string(),
+ ));
+ };
+ let num_values = num_values as usize;
+
+ // The dictionary page is always PLAIN-encoded, regardless of what the
+ // data pages' encoding is (RLE_DICTIONARY/PLAIN_DICTIONARY only describe
+ // how *data* pages reference the dictionary by index).
+ if encoding != Encoding::PLAIN {
+ return Err(ParquetError::General(format!(
+ "Dictionary page encoding must be PLAIN, got {encoding:?}"
+ )));
+ }
+ let mut decoder = ByteArrayDecoderPlain::new(buf, num_values,
Some(num_values), false);
+ let mut offsets = OffsetBuffer::<i32>::with_capacity(num_values);
+ decoder.read(&mut offsets, usize::MAX)?;
+ if offsets.len() != num_values {
+ return Err(ParquetError::General(format!(
+ "Expected {num_values} dictionary values, decoded {}",
+ offsets.len()
+ )));
+ }
+
+ Ok(offsets.into_array(None, ArrowType::Binary))
+}
+
+/// Builds the crypto context needed to decrypt the dictionary page of
+/// `column_metadata`, or `None` if the file (or this column) isn't encrypted.
+#[cfg(feature = "encryption")]
+fn dictionary_page_crypto_context(
+ parquet_meta_data: &ParquetMetaData,
+ column_metadata: &ColumnChunkMetaData,
+ row_group_idx: usize,
+ column_idx: usize,
+) -> Result<Option<Arc<CryptoContext>>> {
+ let Some(file_decryptor) = parquet_meta_data.file_decryptor() else {
+ return Ok(None);
+ };
+ let Some(crypto_metadata) = column_metadata.crypto_metadata() else {
+ return Ok(None);
+ };
+ let ordinal = parquet_meta_data
+ .row_group(row_group_idx)
+ .ordinal()
+ .ok_or_else(|| {
+ ParquetError::General("Encrypted row group is missing its file
ordinal".to_string())
+ })?;
+ let ordinal = usize::try_from(ordinal).map_err(|_| {
+ ParquetError::General("Encrypted row group has an invalid file
ordinal".to_string())
+ })?;
+ let crypto_context =
+ CryptoContext::for_column(file_decryptor, crypto_metadata, ordinal,
column_idx)?
+ .for_dictionary_page();
+ Ok(Some(Arc::new(crypto_context)))
+}
+
+#[cfg(test)]
+mod tests {
+ use super::*;
+ use crate::arrow::ArrowWriter;
+ use crate::basic::Encoding;
+ use crate::file::metadata::ParquetMetaDataReader;
+ use crate::file::properties::WriterProperties;
+ use crate::file::reader::{ChunkReader, FileReader, SerializedFileReader};
+ use crate::parquet_thrift::{ThriftCompactOutputProtocol, WriteThrift};
+ use arrow_array::{Array, BinaryArray, RecordBatch, StringArray};
+ use arrow_schema::{Field, Schema};
+ use std::sync::Arc;
+
+ fn write_dictionary_encoded_strings(values: &[&str]) -> Bytes {
+ let schema = Arc::new(Schema::new(vec![Field::new("s",
ArrowType::Utf8, false)]));
+ let array =
Arc::new(StringArray::from_iter_values(values.iter().copied()));
+ let batch = RecordBatch::try_new(schema.clone(), vec![array]).unwrap();
+
+ let props = WriterProperties::builder()
+ .set_dictionary_enabled(true)
+ .build();
+ let mut buf = Vec::new();
+ {
+ let mut writer = ArrowWriter::try_new(&mut buf, schema,
Some(props)).unwrap();
+ writer.write(&batch).unwrap();
+ writer.close().unwrap();
+ }
+ Bytes::from(buf)
+ }
+
+ #[test]
+ fn decode_dictionary_page_round_trips_strings() {
+ let distinct_values = ["alpha", "beta", "gamma"];
+ // Repeat so the column is worth dictionary-encoding but the
+ // dictionary itself only contains the distinct values.
+ let values: Vec<&str> =
distinct_values.iter().copied().cycle().take(30).collect();
+ let data = write_dictionary_encoded_strings(&values);
+
+ let reader = SerializedFileReader::new(data.clone()).unwrap();
+ let metadata = reader.metadata();
+ let column_metadata = metadata.row_group(0).column(0);
+
+ assert!(
+ column_metadata.dictionary_page_offset().is_some(),
+ "expected the column chunk to be dictionary-encoded"
+ );
+
+ let start = column_metadata.dictionary_page_offset().unwrap() as u64;
+ let end = column_metadata.data_page_offset() as u64;
+ let buffer = data.get_bytes(start, (end - start) as usize).unwrap();
+
+ let array = decode_dictionary_page(buffer, metadata, 0, 0).unwrap();
+ let array = array.as_any().downcast_ref::<BinaryArray>().unwrap();
+ let decoded: Vec<&[u8]> = array.iter().map(|v| v.unwrap()).collect();
+ assert_eq!(decoded, distinct_values.map(str::as_bytes));
+ }
+
+ #[test]
+ fn decode_dictionary_page_errors_on_truncated_buffer() {
+ let distinct_values = ["alpha", "beta", "gamma"];
+ let values: Vec<&str> =
distinct_values.iter().copied().cycle().take(30).collect();
+ let data = write_dictionary_encoded_strings(&values);
+
+ let reader = SerializedFileReader::new(data.clone()).unwrap();
+ let metadata = reader.metadata();
+ let column_metadata = metadata.row_group(0).column(0);
+
+ let start = column_metadata.dictionary_page_offset().unwrap() as u64;
+ let end = column_metadata.data_page_offset() as u64;
+ let buffer = data.get_bytes(start, (end - start) as usize).unwrap();
+
+ // Simulate a truncated/malformed file: the page header's declared
+ // `compressed_page_size` no longer fits in what was actually
+ // fetched. This must return an error rather than panic while
+ // slicing (`Bytes::slice` panics on out-of-bounds ranges).
+ let truncated = buffer.slice(..buffer.len() - 1);
+ let err = decode_dictionary_page(truncated, metadata, 0,
0).unwrap_err();
+ assert!(
+ matches!(err, ParquetError::EOF(_)),
+ "unexpected error: {err}"
+ );
+ }
+
+ fn dictionary_page_with_header_change(
+ data: &Bytes,
+ change: impl FnOnce(&mut crate::file::metadata::thrift::PageHeader),
+ ) -> (Bytes, ParquetMetaData) {
+ let reader = SerializedFileReader::new(data.clone()).unwrap();
+ let metadata = reader.metadata().clone();
+ let column = metadata.row_group(0).column(0);
+ let start = column.dictionary_page_offset().unwrap() as u64;
+ let end = column.data_page_offset() as u64;
+ let buffer = data.get_bytes(start, (end - start) as usize).unwrap();
+ let context = SerializedPageReaderContext {
+ read_stats: true,
+ #[cfg(feature = "encryption")]
+ crypto_context: None,
+ };
+ let (header_len, mut header) =
+ read_page_header_len_from_bytes(&context, &buffer, 0,
true).unwrap();
+ change(&mut header);
+ let mut changed = Vec::new();
+ header
+ .write_thrift(&mut ThriftCompactOutputProtocol::new(&mut changed))
+ .unwrap();
+ changed.extend_from_slice(&buffer[header_len..]);
+ (Bytes::from(changed), metadata)
+ }
+
+ #[test]
+ fn decode_dictionary_page_rejects_missing_values() {
+ let data = write_dictionary_encoded_strings(&["alpha", "beta",
"alpha"]);
+ let (buffer, metadata) = dictionary_page_with_header_change(&data,
|header| {
+ header.dictionary_page_header.as_mut().unwrap().num_values += 1;
+ });
+ let err = decode_dictionary_page(buffer, &metadata, 0, 0).unwrap_err();
+ assert!(err.to_string().contains("dictionary values"), "{err}");
+ }
+
+ #[test]
+ fn decode_dictionary_page_rejects_non_plain_encoding() {
+ let data = write_dictionary_encoded_strings(&["alpha", "beta",
"alpha"]);
+ let (buffer, metadata) = dictionary_page_with_header_change(&data,
|header| {
+ header.dictionary_page_header.as_mut().unwrap().encoding =
Encoding::RLE_DICTIONARY;
+ });
+ let err = decode_dictionary_page(buffer, &metadata, 0, 0).unwrap_err();
+ assert!(err.to_string().contains("PLAIN"), "{err}");
+ }
+
+ #[test]
+ fn decode_dictionary_page_rejects_non_byte_array() {
+ let schema = Arc::new(Schema::new(vec![Field::new("i",
ArrowType::Int32, false)]));
+ let array = Arc::new(arrow_array::Int32Array::from(vec![1, 2, 3]));
+ let batch = RecordBatch::try_new(schema.clone(), vec![array]).unwrap();
+ let mut buf = Vec::new();
+ {
+ let mut writer = ArrowWriter::try_new(&mut buf, schema,
None).unwrap();
+ writer.write(&batch).unwrap();
+ writer.close().unwrap();
+ }
+ let data = Bytes::from(buf);
+
+ let reader = SerializedFileReader::new(data).unwrap();
+ let metadata = reader.metadata();
+
+ let err = decode_dictionary_page(Bytes::new(), metadata, 0,
0).unwrap_err();
+ assert!(err.to_string().contains("BYTE_ARRAY"));
+ }
+
+ #[test]
+ fn read_column_dictionary_round_trips_via_metadata_reader() {
+ let distinct_values = ["alpha", "beta", "gamma"];
+ let values: Vec<&str> =
distinct_values.iter().copied().cycle().take(30).collect();
+ let data = write_dictionary_encoded_strings(&values);
+
+ let reader = SerializedFileReader::new(data.clone()).unwrap();
+ let metadata = reader.metadata();
+
+ let array = ParquetMetaDataReader::read_column_dictionary(&data,
metadata, 0, 0)
+ .unwrap()
+ .unwrap();
+ let array = array.as_any().downcast_ref::<BinaryArray>().unwrap();
+ let decoded: Vec<&[u8]> = array.iter().map(|v| v.unwrap()).collect();
+ assert_eq!(decoded, distinct_values.map(str::as_bytes));
+ }
+
+ #[cfg(feature = "encryption")]
+ #[test]
+ fn read_column_dictionary_round_trips_with_encryption() {
+ use crate::encryption::decrypt::FileDecryptionProperties;
+ use crate::encryption::encrypt::FileEncryptionProperties;
+ const FOOTER_KEY: &[u8] = b"0123456789012345";
+
+ let distinct_values = ["alpha", "beta", "gamma"];
+ let values: Vec<&str> =
distinct_values.iter().copied().cycle().take(30).collect();
+
+ let schema = Arc::new(Schema::new(vec![Field::new("s",
ArrowType::Utf8, false)]));
+ let array =
Arc::new(StringArray::from_iter_values(values.iter().copied()));
+ let batch = RecordBatch::try_new(schema.clone(), vec![array]).unwrap();
+
+ let encryption_properties =
FileEncryptionProperties::builder(FOOTER_KEY.to_vec())
+ .build()
+ .unwrap();
+ let props = WriterProperties::builder()
+ .set_dictionary_enabled(true)
+ .with_file_encryption_properties(encryption_properties)
+ .build();
+ let mut buf = Vec::new();
+ {
+ let mut writer = ArrowWriter::try_new(&mut buf, schema,
Some(props)).unwrap();
+ writer.write(&batch).unwrap();
+ writer.close().unwrap();
+ }
+ let data = Bytes::from(buf);
+
+ let decryption_properties =
FileDecryptionProperties::builder(FOOTER_KEY.to_vec())
+ .build()
+ .unwrap();
+ let metadata = ParquetMetaDataReader::new()
+ .with_decryption_properties(Some(decryption_properties))
+ .parse_and_finish(&data)
+ .unwrap();
+
+ let array = ParquetMetaDataReader::read_column_dictionary(&data,
&metadata, 0, 0)
+ .unwrap()
+ .unwrap();
+ let array = array.as_any().downcast_ref::<BinaryArray>().unwrap();
+ let decoded: Vec<&[u8]> = array.iter().map(|v| v.unwrap()).collect();
+ assert_eq!(decoded, distinct_values.map(str::as_bytes));
+ }
+
+ #[cfg(feature = "encryption")]
+ #[test]
+ fn read_column_dictionary_uses_file_ordinal_after_filtering() {
+ use crate::encryption::decrypt::FileDecryptionProperties;
+ use crate::encryption::encrypt::FileEncryptionProperties;
+ use crate::file::metadata::ParquetMetaDataBuilder;
+
+ const FOOTER_KEY: &[u8] = b"0123456789012345";
+ let schema = Arc::new(Schema::new(vec![Field::new("s",
ArrowType::Utf8, false)]));
+ let props = WriterProperties::builder()
+ .set_dictionary_enabled(true)
+ .with_file_encryption_properties(
+ FileEncryptionProperties::builder(FOOTER_KEY.to_vec())
+ .build()
+ .unwrap(),
+ )
+ .build();
+ let mut buf = Vec::new();
+ {
+ let mut writer = ArrowWriter::try_new(&mut buf, schema.clone(),
Some(props)).unwrap();
+ for value in ["first", "second"] {
+ let array =
Arc::new(StringArray::from_iter_values(std::iter::repeat_n(
+ value, 30,
+ )));
+ let batch = RecordBatch::try_new(schema.clone(),
vec![array]).unwrap();
+ writer.write(&batch).unwrap();
+ writer.flush().unwrap();
+ }
+ writer.close().unwrap();
+ }
+ let data = Bytes::from(buf);
+ let metadata = ParquetMetaDataReader::new()
+ .with_decryption_properties(Some(
+ FileDecryptionProperties::builder(FOOTER_KEY.to_vec())
+ .build()
+ .unwrap(),
+ ))
+ .parse_and_finish(&data)
+ .unwrap();
+ assert_eq!(metadata.row_group(1).ordinal(), Some(1));
+ let second = metadata.row_group(1).clone();
+ let filtered = ParquetMetaDataBuilder::new_from_metadata(metadata)
+ .set_row_groups(vec![second])
+ .build();
+ let array = ParquetMetaDataReader::read_column_dictionary(&data,
&filtered, 0, 0)
+ .unwrap()
+ .unwrap();
+ let array = array.as_any().downcast_ref::<BinaryArray>().unwrap();
+ assert_eq!(array.value(0), b"second");
+
+ let invalid = filtered
+ .row_group(0)
+ .clone()
+ .into_builder()
+ .set_ordinal(-1)
+ .build()
+ .unwrap();
+ let invalid_metadata =
ParquetMetaDataBuilder::new_from_metadata(filtered)
+ .set_row_groups(vec![invalid])
+ .build();
+ let err = ParquetMetaDataReader::read_column_dictionary(&data,
&invalid_metadata, 0, 0)
+ .unwrap_err();
+ assert!(err.to_string().contains("invalid file ordinal"), "{err}");
+ }
+
+ #[test]
+ fn read_column_dictionary_returns_none_without_dictionary_page() {
+ use crate::file::metadata::ParquetMetaDataReader;
+
+ let schema = Arc::new(Schema::new(vec![Field::new("s",
ArrowType::Utf8, false)]));
+ let array = Arc::new(StringArray::from_iter_values(["a", "b", "c"]));
+ let batch = RecordBatch::try_new(schema.clone(), vec![array]).unwrap();
+
+ let props = WriterProperties::builder()
+ .set_dictionary_enabled(false)
+ .build();
+ let mut buf = Vec::new();
+ {
+ let mut writer = ArrowWriter::try_new(&mut buf, schema,
Some(props)).unwrap();
+ writer.write(&batch).unwrap();
+ writer.close().unwrap();
+ }
+ let data = Bytes::from(buf);
+
+ let reader = SerializedFileReader::new(data.clone()).unwrap();
+ let metadata = reader.metadata();
+ assert!(
+ metadata
+ .row_group(0)
+ .column(0)
+ .dictionary_page_offset()
+ .is_none()
+ );
+
+ let result = ParquetMetaDataReader::read_column_dictionary(&data,
metadata, 0, 0).unwrap();
+ assert!(result.is_none());
+ }
+}
diff --git a/parquet/src/file/metadata/mod.rs b/parquet/src/file/metadata/mod.rs
index 166a7eb7ab..a12fbdd1de 100644
--- a/parquet/src/file/metadata/mod.rs
+++ b/parquet/src/file/metadata/mod.rs
@@ -49,6 +49,9 @@
//! Please see [`external_metadata.rs`]
//!
//! [`external_metadata.rs`]:
https://github.com/apache/arrow-rs/tree/master/parquet/examples/external_metadata.rs
+//!
+#[cfg(feature = "arrow")]
+mod dictionary;
mod footer_tail;
mod memory;
mod options;
diff --git a/parquet/src/file/metadata/reader.rs
b/parquet/src/file/metadata/reader.rs
index 3ffce28523..24bc5c982b 100644
--- a/parquet/src/file/metadata/reader.rs
+++ b/parquet/src/file/metadata/reader.rs
@@ -19,6 +19,8 @@
use crate::encryption::decrypt::FileDecryptionProperties;
use crate::errors::{ParquetError, Result};
use crate::file::FOOTER_SIZE;
+#[cfg(feature = "arrow")]
+use crate::file::metadata::dictionary::decode_dictionary_page;
use crate::file::metadata::parser::decode_metadata;
use crate::file::metadata::thrift::parquet_schema_from_bytes;
use crate::file::metadata::{
@@ -26,6 +28,8 @@ use crate::file::metadata::{
};
use crate::file::reader::ChunkReader;
use crate::schema::types::SchemaDescriptor;
+#[cfg(feature = "arrow")]
+use arrow_array::ArrayRef;
use bytes::Bytes;
use std::sync::Arc;
use std::{io::Read, ops::Range};
@@ -472,6 +476,53 @@ impl ParquetMetaDataReader {
self.load_page_index_with_remainder(fetch, None).await
}
+ /// Reads and decodes the dictionary page of a column chunk into an Arrow
array.
+ ///
+ /// Returns `Ok(None)` if the column chunk has no dictionary page, or if
+ /// its physical type is not `BYTE_ARRAY` (the only physical type
+ /// currently supported).
+ ///
+ /// The returned array contains raw `Binary` values, even for columns
+ /// annotated as strings. Callers can compare byte slices directly or
+ /// convert values to UTF-8 explicitly.
+ ///
+ /// This can be used to inspect dictionary values when selecting or pruning
+ /// row groups before reading their data pages.
+ ///
+ /// Note this does not verify that the *entire* column chunk is
+ /// dictionary-encoded (i.e. that the dictionary contains every value in
+ /// the chunk) -- callers that need that guarantee should check
+ ///
[`crate::file::metadata::ColumnChunkMetaData::page_encoding_stats_mask`].
+ #[cfg(feature = "arrow")]
+ pub fn read_column_dictionary<R: ChunkReader>(
+ reader: &R,
+ metadata: &ParquetMetaData,
+ row_group_idx: usize,
+ column_idx: usize,
+ ) -> Result<Option<ArrayRef>> {
+ let Some(range) = dictionary_page_byte_range(metadata, row_group_idx,
column_idx)? else {
+ return Ok(None);
+ };
+ let length = usize::try_from(range.end - range.start)?;
+ let buffer = reader.get_bytes(range.start, length)?;
+ decode_dictionary_page(buffer, metadata, row_group_idx,
column_idx).map(Some)
+ }
+
+ /// Asynchronous version of [`Self::read_column_dictionary`].
+ #[cfg(all(feature = "async", feature = "arrow"))]
+ pub async fn read_column_dictionary_async<F: MetadataFetch>(
+ mut fetch: F,
+ metadata: &ParquetMetaData,
+ row_group_idx: usize,
+ column_idx: usize,
+ ) -> Result<Option<ArrayRef>> {
+ let Some(range) = dictionary_page_byte_range(metadata, row_group_idx,
column_idx)? else {
+ return Ok(None);
+ };
+ let buffer = fetch.fetch(range).await?;
+ decode_dictionary_page(buffer, metadata, row_group_idx,
column_idx).map(Some)
+ }
+
#[cfg(all(feature = "async", feature = "arrow"))]
async fn load_page_index_with_remainder<F: MetadataFetch>(
&mut self,
@@ -841,6 +892,41 @@ fn parse_index_data(push_decoder: &mut
ParquetMetaDataPushDecoder) -> Result<Par
}
}
+/// Returns the `[start, end)` byte range of a column chunk's dictionary
+/// page, or `None` if the column chunk has no dictionary page or is not a
+/// `BYTE_ARRAY` column (the only physical type [`decode_dictionary_page`]
+/// currently supports).
+#[cfg(feature = "arrow")]
+fn dictionary_page_byte_range(
+ metadata: &ParquetMetaData,
+ row_group_idx: usize,
+ column_idx: usize,
+) -> Result<Option<Range<u64>>> {
+ let column_metadata = metadata.row_group(row_group_idx).column(column_idx);
+ let column_descriptor = column_metadata.column_descr();
+
+ if column_descriptor.physical_type() != crate::basic::Type::BYTE_ARRAY {
+ return Ok(None);
+ }
+ let Some(start) = column_metadata.dictionary_page_offset() else {
+ return Ok(None);
+ };
+ let start: u64 = start
+ .try_into()
+ .map_err(|_| ParquetError::General("Dictionary page offset is
invalid".to_string()))?;
+ let end: u64 = column_metadata
+ .data_page_offset()
+ .try_into()
+ .map_err(|_| ParquetError::General("Data page offset is
invalid".to_string()))?;
+ if end < start {
+ return Err(ParquetError::General(
+ "Data page offset precedes dictionary page offset".to_string(),
+ ));
+ }
+
+ Ok(Some(start..end))
+}
+
#[cfg(test)]
mod tests {
use super::*;
diff --git a/parquet/src/file/serialized_reader.rs
b/parquet/src/file/serialized_reader.rs
index 9f411b2dc9..7345fb9645 100644
--- a/parquet/src/file/serialized_reader.rs
+++ b/parquet/src/file/serialized_reader.rs
@@ -563,12 +563,12 @@ enum SerializedPageReaderState {
}
#[derive(Default)]
-struct SerializedPageReaderContext {
+pub(crate) struct SerializedPageReaderContext {
/// Controls decoding of page-level statistics
- read_stats: bool,
+ pub(crate) read_stats: bool,
/// Crypto context carrying objects required for decryption
#[cfg(feature = "encryption")]
- crypto_context: Option<Arc<CryptoContext>>,
+ pub(crate) crypto_context: Option<Arc<CryptoContext>>,
}
/// A serialized implementation for Parquet [`PageReader`].
@@ -781,23 +781,30 @@ impl<R: ChunkReader> SerializedPageReader<R> {
let header = context.read_page_header(&mut tracked, page_index,
dictionary_page)?;
Ok((tracked.bytes_read, header))
}
+}
- fn read_page_header_len_from_bytes(
- context: &SerializedPageReaderContext,
- buffer: &[u8],
- page_index: usize,
- dictionary_page: bool,
- ) -> Result<(usize, PageHeader)> {
- let mut input = std::io::Cursor::new(buffer);
- let header = context.read_page_header(&mut input, page_index,
dictionary_page)?;
- let header_len = input.position() as usize;
- Ok((header_len, header))
- }
+/// Reads (and decrypts, if `context` carries a crypto context) the page
header stored
+/// at the front of `buffer`, returning the header and the number of bytes of
`buffer`
+/// it occupies.
+///
+/// This is exposed for callers that need to decode a single page directly
from an
+/// already-fetched byte range, outside of the normal [`SerializedPageReader`]
iteration
+/// -- e.g. decoding a dictionary page standalone.
+pub(crate) fn read_page_header_len_from_bytes(
+ context: &SerializedPageReaderContext,
+ buffer: &[u8],
+ page_index: usize,
+ dictionary_page: bool,
+) -> Result<(usize, PageHeader)> {
+ let mut input = std::io::Cursor::new(buffer);
+ let header = context.read_page_header(&mut input, page_index,
dictionary_page)?;
+ let header_len = input.position() as usize;
+ Ok((header_len, header))
}
#[cfg(not(feature = "encryption"))]
impl SerializedPageReaderContext {
- fn read_page_header<T: Read>(
+ pub(crate) fn read_page_header<T: Read>(
&self,
input: &mut T,
_page_index: usize,
@@ -811,7 +818,7 @@ impl SerializedPageReaderContext {
}
}
- fn decrypt_page_data<T>(
+ pub(crate) fn decrypt_page_data<T>(
&self,
buffer: T,
_page_index: usize,
@@ -823,7 +830,7 @@ impl SerializedPageReaderContext {
#[cfg(feature = "encryption")]
impl SerializedPageReaderContext {
- fn read_page_header<T: Read>(
+ pub(crate) fn read_page_header<T: Read>(
&self,
input: &mut T,
page_index: usize,
@@ -861,7 +868,12 @@ impl SerializedPageReaderContext {
}
}
- fn decrypt_page_data<T>(&self, buffer: T, page_index: usize,
dictionary_page: bool) -> Result<T>
+ pub(crate) fn decrypt_page_data<T>(
+ &self,
+ buffer: T,
+ page_index: usize,
+ dictionary_page: bool,
+ ) -> Result<T>
where
T: AsRef<[u8]>,
T: From<Vec<u8>>,
@@ -907,7 +919,7 @@ fn verify_page_header_len(header_len: usize,
remaining_bytes: u64) -> Result<()>
Ok(())
}
-fn verify_page_size(
+pub(crate) fn verify_page_size(
compressed_size: i32,
uncompressed_size: i32,
remaining_bytes: u64,
@@ -1001,7 +1013,7 @@ impl<R: ChunkReader> PageReader for
SerializedPageReader<R> {
let page_len =
usize::try_from(front.compressed_page_size)?;
let buffer = self.reader.get_bytes(front.offset as u64,
page_len)?;
- let (offset, header) =
Self::read_page_header_len_from_bytes(
+ let (offset, header) = read_page_header_len_from_bytes(
&self.context,
buffer.as_ref(),
*page_index,