alamb commented on code in PR #10555:
URL: https://github.com/apache/arrow-rs/pull/10555#discussion_r4160254352


##########
parquet/benches/push_decoder.rs:
##########
@@ -166,5 +173,106 @@ fn bench_nbuf(c: &mut Criterion) {
     group.finish();
 }
 
-criterion_group!(benches, bench_1buf, bench_nbuf);
+/// Write a Parquet file with `num_columns` columns and one row group of
+/// 1,000 rows, with `pages_per_column` data pages per column chunk.
+fn make_paged_test_file(num_columns: usize, pages_per_column: usize) -> Bytes {
+    let num_rows = 1_000;
+    let schema = make_wide_schema(num_columns);
+    let columns: Vec<Arc<dyn arrow_array::Array>> = (0..num_columns)
+        .map(|_| {
+            Arc::new(Float32Array::from_iter_values(
+                (0..num_rows).map(|v| v as f32),
+            )) as _
+        })
+        .collect();
+    let batch = RecordBatch::try_new(schema.clone(), columns).unwrap();
+
+    let page_rows = num_rows / pages_per_column;
+    let mut buf = Vec::new();
+    let props = WriterProperties::builder()
+        .set_max_row_group_row_count(Some(num_rows))
+        .set_data_page_row_count_limit(page_rows)
+        // Page limits are checked between write batches.
+        .set_write_batch_size(page_rows)
+        .set_dictionary_enabled(false)
+        .build();
+    let mut writer = ArrowWriter::try_new(&mut buf, schema, 
Some(props)).unwrap();
+    writer.write(&batch).unwrap();
+    writer.close().unwrap();
+    Bytes::from(buf)
+}
+
+type BuilderFn<'a> = Box<dyn Fn() -> ParquetPushDecoderBuilder + 'a>;
+
+/// Plan a wide file with a page index: the first range (what a read-ahead
+/// caller waits for) and the whole scan, which is one row group. For
+/// comparison, `first_reader` builds a decoder, pushes the whole file and
+/// builds the first row group reader, which includes the decoder's own
+/// per-row-group setup.
+///
+/// `all` reads every column and row. `selection` keeps 10 rows of every 100,
+/// so each column chunk reads every other page. `narrow` reads 10 columns.
+fn bench_scan_plan(c: &mut Criterion) {
+    let mut group = c.benchmark_group("push_decoder/scan_plan");
+
+    for (num_cols, pages) in [(100, 10), (1_000, 10), (10_000, 10), (1_000, 
100)] {
+        let file_data = make_paged_test_file(num_cols, pages);
+        let options = 
ArrowReaderOptions::new().with_page_index_policy(PageIndexPolicy::Required);
+        let metadata = ArrowReaderMetadata::load(&file_data, options).unwrap();
+        let selection = RowSelection::from(
+            (0..10)
+                .flat_map(|_| [RowSelector::select(10), RowSelector::skip(90)])
+                .collect::<Vec<_>>(),
+        );
+        let narrow = ProjectionMask::leaves(metadata.parquet_schema(), 0..10);
+        let variants: [(&str, BuilderFn); 3] = [
+            (
+                "all",
+                Box::new(|| 
ParquetPushDecoderBuilder::new_with_metadata(metadata.clone())),

Review Comment:
   do you need the function? I think cloning an existing builder would be 
simpler and easier to understand



##########
parquet/benches/push_decoder.rs:
##########
@@ -166,5 +173,106 @@ fn bench_nbuf(c: &mut Criterion) {
     group.finish();
 }
 
-criterion_group!(benches, bench_1buf, bench_nbuf);
+/// Write a Parquet file with `num_columns` columns and one row group of
+/// 1,000 rows, with `pages_per_column` data pages per column chunk.
+fn make_paged_test_file(num_columns: usize, pages_per_column: usize) -> Bytes {
+    let num_rows = 1_000;
+    let schema = make_wide_schema(num_columns);
+    let columns: Vec<Arc<dyn arrow_array::Array>> = (0..num_columns)
+        .map(|_| {
+            Arc::new(Float32Array::from_iter_values(
+                (0..num_rows).map(|v| v as f32),
+            )) as _
+        })
+        .collect();
+    let batch = RecordBatch::try_new(schema.clone(), columns).unwrap();
+
+    let page_rows = num_rows / pages_per_column;
+    let mut buf = Vec::new();
+    let props = WriterProperties::builder()
+        .set_max_row_group_row_count(Some(num_rows))
+        .set_data_page_row_count_limit(page_rows)
+        // Page limits are checked between write batches.
+        .set_write_batch_size(page_rows)
+        .set_dictionary_enabled(false)
+        .build();
+    let mut writer = ArrowWriter::try_new(&mut buf, schema, 
Some(props)).unwrap();
+    writer.write(&batch).unwrap();
+    writer.close().unwrap();
+    Bytes::from(buf)
+}
+
+type BuilderFn<'a> = Box<dyn Fn() -> ParquetPushDecoderBuilder + 'a>;
+
+/// Plan a wide file with a page index: the first range (what a read-ahead
+/// caller waits for) and the whole scan, which is one row group. For
+/// comparison, `first_reader` builds a decoder, pushes the whole file and
+/// builds the first row group reader, which includes the decoder's own
+/// per-row-group setup.
+///
+/// `all` reads every column and row. `selection` keeps 10 rows of every 100,
+/// so each column chunk reads every other page. `narrow` reads 10 columns.
+fn bench_scan_plan(c: &mut Criterion) {
+    let mut group = c.benchmark_group("push_decoder/scan_plan");
+
+    for (num_cols, pages) in [(100, 10), (1_000, 10), (10_000, 10), (1_000, 
100)] {
+        let file_data = make_paged_test_file(num_cols, pages);
+        let options = 
ArrowReaderOptions::new().with_page_index_policy(PageIndexPolicy::Required);
+        let metadata = ArrowReaderMetadata::load(&file_data, options).unwrap();
+        let selection = RowSelection::from(
+            (0..10)
+                .flat_map(|_| [RowSelector::select(10), RowSelector::skip(90)])
+                .collect::<Vec<_>>(),
+        );
+        let narrow = ProjectionMask::leaves(metadata.parquet_schema(), 0..10);
+        let variants: [(&str, BuilderFn); 3] = [
+            (
+                "all",
+                Box::new(|| 
ParquetPushDecoderBuilder::new_with_metadata(metadata.clone())),
+            ),
+            (
+                "selection",
+                Box::new(|| {
+                    
ParquetPushDecoderBuilder::new_with_metadata(metadata.clone())
+                        .with_row_selection(selection.clone())
+                }),
+            ),
+            (
+                "narrow",
+                Box::new(|| {
+                    
ParquetPushDecoderBuilder::new_with_metadata(metadata.clone())
+                        .with_projection(narrow.clone())
+                }),
+            ),
+        ];
+
+        for (variant, builder) in &variants {
+            let decoder = builder().build().unwrap();
+            let id = format!("{variant}/{num_cols}cols_{pages}pages");
+
+            group.bench_function(BenchmarkId::new("first_range", &id), |b| {
+                b.iter(|| black_box(decoder.scan_plan().next()))
+            });
+            group.bench_function(BenchmarkId::new("whole_scan", &id), |b| {
+                b.iter(|| black_box(decoder.scan_plan().count()))

Review Comment:
   maybe it is worth a comment here that the scan plan is lazily evaluated (so 
counting forces it to be evaluated)?



##########
parquet/src/arrow/push_decoder/reader_builder/mod.rs:
##########
@@ -822,6 +753,14 @@ impl RowGroupReaderBuilder {
         Ok(result)
     }
 
+    /// A [`ScanPlanBuilder`] that plans the same ranges as this builder, for

Review Comment:
   could the same thing be accomplished via `self.clone().into_builder()` 
instead of a new API? Since this is crate local I think it is fine, I am just 
asking



##########
parquet/benches/push_decoder.rs:
##########
@@ -166,5 +173,106 @@ fn bench_nbuf(c: &mut Criterion) {
     group.finish();
 }
 
-criterion_group!(benches, bench_1buf, bench_nbuf);
+/// Write a Parquet file with `num_columns` columns and one row group of
+/// 1,000 rows, with `pages_per_column` data pages per column chunk.
+fn make_paged_test_file(num_columns: usize, pages_per_column: usize) -> Bytes {
+    let num_rows = 1_000;
+    let schema = make_wide_schema(num_columns);
+    let columns: Vec<Arc<dyn arrow_array::Array>> = (0..num_columns)
+        .map(|_| {
+            Arc::new(Float32Array::from_iter_values(
+                (0..num_rows).map(|v| v as f32),
+            )) as _
+        })
+        .collect();
+    let batch = RecordBatch::try_new(schema.clone(), columns).unwrap();
+
+    let page_rows = num_rows / pages_per_column;
+    let mut buf = Vec::new();
+    let props = WriterProperties::builder()
+        .set_max_row_group_row_count(Some(num_rows))
+        .set_data_page_row_count_limit(page_rows)
+        // Page limits are checked between write batches.
+        .set_write_batch_size(page_rows)
+        .set_dictionary_enabled(false)
+        .build();
+    let mut writer = ArrowWriter::try_new(&mut buf, schema, 
Some(props)).unwrap();
+    writer.write(&batch).unwrap();
+    writer.close().unwrap();
+    Bytes::from(buf)
+}
+
+type BuilderFn<'a> = Box<dyn Fn() -> ParquetPushDecoderBuilder + 'a>;
+
+/// Plan a wide file with a page index: the first range (what a read-ahead
+/// caller waits for) and the whole scan, which is one row group. For
+/// comparison, `first_reader` builds a decoder, pushes the whole file and
+/// builds the first row group reader, which includes the decoder's own
+/// per-row-group setup.
+///
+/// `all` reads every column and row. `selection` keeps 10 rows of every 100,
+/// so each column chunk reads every other page. `narrow` reads 10 columns.
+fn bench_scan_plan(c: &mut Criterion) {
+    let mut group = c.benchmark_group("push_decoder/scan_plan");
+
+    for (num_cols, pages) in [(100, 10), (1_000, 10), (10_000, 10), (1_000, 
100)] {
+        let file_data = make_paged_test_file(num_cols, pages);
+        let options = 
ArrowReaderOptions::new().with_page_index_policy(PageIndexPolicy::Required);
+        let metadata = ArrowReaderMetadata::load(&file_data, options).unwrap();
+        let selection = RowSelection::from(
+            (0..10)
+                .flat_map(|_| [RowSelector::select(10), RowSelector::skip(90)])
+                .collect::<Vec<_>>(),
+        );
+        let narrow = ProjectionMask::leaves(metadata.parquet_schema(), 0..10);
+        let variants: [(&str, BuilderFn); 3] = [
+            (
+                "all",
+                Box::new(|| 
ParquetPushDecoderBuilder::new_with_metadata(metadata.clone())),
+            ),
+            (
+                "selection",
+                Box::new(|| {
+                    
ParquetPushDecoderBuilder::new_with_metadata(metadata.clone())
+                        .with_row_selection(selection.clone())
+                }),
+            ),
+            (
+                "narrow",
+                Box::new(|| {
+                    
ParquetPushDecoderBuilder::new_with_metadata(metadata.clone())
+                        .with_projection(narrow.clone())
+                }),
+            ),
+        ];
+
+        for (variant, builder) in &variants {
+            let decoder = builder().build().unwrap();
+            let id = format!("{variant}/{num_cols}cols_{pages}pages");
+
+            group.bench_function(BenchmarkId::new("first_range", &id), |b| {
+                b.iter(|| black_box(decoder.scan_plan().next()))
+            });
+            group.bench_function(BenchmarkId::new("whole_scan", &id), |b| {
+                b.iter(|| black_box(decoder.scan_plan().count()))
+            });
+            group.bench_function(BenchmarkId::new("first_reader", &id), |b| {
+                b.iter(|| {

Review Comment:
   this benchmark seems to have nothing to do with scan_plan. Why included it?



##########
parquet/src/arrow/push_decoder/scan_plan/frontier.rs:
##########
@@ -0,0 +1,494 @@
+// Licensed to the Apache Software Foundation (ASF) under one
+// or more contributor license agreements.  See the NOTICE file
+// distributed with this work for additional information
+// regarding copyright ownership.  The ASF licenses this file
+// to you under the Apache License, Version 2.0 (the
+// "License"); you may not use this file except in compliance
+// with the License.  You may obtain a copy of the License at
+//
+//   http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing,
+// software distributed under the License is distributed on an
+// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+// KIND, either express or implied.  See the License for the
+// specific language governing permissions and limitations
+// under the License.
+
+//! Which row groups a scan reads next, with their selections and budget.
+
+use crate::arrow::arrow_reader::{RowGroupPlan, RowGroupSelection, 
RowSelection};
+use crate::errors::ParquetError;
+use crate::file::metadata::ParquetMetaData;
+use std::collections::VecDeque;
+use std::sync::Arc;
+
+use super::budget::RowBudget;
+
+/// Plan for the next queued row group after row-selection slicing.
+#[derive(Debug)]
+enum QueuedRowGroupDecision {
+    /// Hand this row group to the builder.
+    Read(NextRowGroup),
+    /// Skip this row group, and keep scanning with the updated budget.
+    Skip { remaining_budget: RowBudget },
+}
+
+/// Work item handed from [`RowGroupFrontier`] to 
[`RowGroupReaderBuilder`](crate::arrow::push_decoder::reader_builder::RowGroupReaderBuilder).
+#[derive(Debug, Clone)]
+pub(crate) struct NextRowGroup {

Review Comment:
   these are just moved to a new file as I undersand



##########
parquet/src/arrow/push_decoder/scan_plan/mod.rs:
##########
@@ -0,0 +1,1778 @@
+// Licensed to the Apache Software Foundation (ASF) under one
+// or more contributor license agreements.  See the NOTICE file
+// distributed with this work for additional information
+// regarding copyright ownership.  The ASF licenses this file
+// to you under the Apache License, Version 2.0 (the
+// "License"); you may not use this file except in compliance
+// with the License.  You may obtain a copy of the License at
+//
+//   http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing,
+// software distributed under the License is distributed on an
+// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+// KIND, either express or implied.  See the License for the
+// specific language governing permissions and limitations
+// under the License.
+
+//! [`ScanPlan`]: the byte ranges that a push decoder may still read.
+
+use std::cmp::Reverse;
+use std::collections::BinaryHeap;
+use std::iter::FusedIterator;
+use std::ops::Range;
+use std::sync::Arc;
+
+use crate::arrow::ProjectionMask;
+use crate::arrow::arrow_reader::{ReadPlanBuilder, RowSelection};
+use crate::errors::ParquetError;
+use crate::file::metadata::page_index::PageIndexProvider;
+use crate::file::page_index::offset_index::PageLocation;
+
+mod budget;
+mod frontier;
+
+pub(crate) use budget::{BudgetedReadPlan, RowBudget};
+pub(crate) use frontier::{NextRowGroup, RowGroupFrontier};
+
+/// One item of a [`ScanPlan`].
+#[derive(Debug, Clone, PartialEq, Eq)]
+#[non_exhaustive]
+pub struct PlannedRange {
+    /// Byte range in the file.
+    pub range: Range<u64>,
+    /// Row group index in the file.
+    pub row_group: usize,
+    /// Leaf column index in the file.
+    pub column: usize,
+    /// What the range contains.
+    pub kind: PageKind,
+    /// First planned row that this range serves, counted from the start of
+    /// the plan. Used only to order the plan.
+    pub(crate) first_row: u64,
+    /// One past the last planned row that this range serves.
+    pub(crate) last_row: u64,
+    /// The decoding stage that first reads this range.
+    pub(crate) stage: ScanStage,
+    /// `true` if a predicate result can make this range unnecessary.
+    pub(crate) conditional: bool,
+}
+
+impl PlannedRange {
+    /// Number of bytes in this range.
+    pub fn len(&self) -> u64 {
+        self.range.end - self.range.start
+    }
+
+    /// Returns `true` if this range contains no bytes.
+    pub fn is_empty(&self) -> bool {
+        self.range.is_empty()
+    }
+}
+
+/// What a [`PlannedRange`] contains.
+#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
+#[non_exhaustive]
+pub enum PageKind {
+    /// The dictionary page of a column chunk.
+    Dictionary,
+    /// One data page.
+    Data,
+    /// A complete column chunk. The plan uses this when the column has no
+    /// offset index, so page locations are unknown.
+    ColumnChunk,
+}
+
+/// The decoding stage that first reads a [`PlannedRange`].
+///
+/// The decoder reads a column once per row group. A column that a predicate
+/// reads is not read again for the output.
+#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
+pub(crate) enum ScanStage {
+    /// Evaluation of the predicate at this index in the
+    /// [`RowFilter`](crate::arrow::arrow_reader::RowFilter).
+    Predicate(usize),
+    /// Decoding of the output columns.
+    Projection,
+}
+
+/// The byte ranges that a push decoder may still read, in the order that
+/// decoding needs them.
+///
+/// Created by [`ParquetPushDecoder::scan_plan`]. Use it to fetch data before

Review Comment:
   👍 



##########
parquet/src/arrow/push_decoder/scan_plan/mod.rs:
##########
@@ -0,0 +1,1778 @@
+// Licensed to the Apache Software Foundation (ASF) under one
+// or more contributor license agreements.  See the NOTICE file
+// distributed with this work for additional information
+// regarding copyright ownership.  The ASF licenses this file
+// to you under the Apache License, Version 2.0 (the
+// "License"); you may not use this file except in compliance
+// with the License.  You may obtain a copy of the License at
+//
+//   http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing,
+// software distributed under the License is distributed on an
+// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+// KIND, either express or implied.  See the License for the
+// specific language governing permissions and limitations
+// under the License.
+
+//! [`ScanPlan`]: the byte ranges that a push decoder may still read.
+
+use std::cmp::Reverse;
+use std::collections::BinaryHeap;
+use std::iter::FusedIterator;
+use std::ops::Range;
+use std::sync::Arc;
+
+use crate::arrow::ProjectionMask;
+use crate::arrow::arrow_reader::{ReadPlanBuilder, RowSelection};
+use crate::errors::ParquetError;
+use crate::file::metadata::page_index::PageIndexProvider;
+use crate::file::page_index::offset_index::PageLocation;
+
+mod budget;
+mod frontier;
+
+pub(crate) use budget::{BudgetedReadPlan, RowBudget};
+pub(crate) use frontier::{NextRowGroup, RowGroupFrontier};
+
+/// One item of a [`ScanPlan`].
+#[derive(Debug, Clone, PartialEq, Eq)]
+#[non_exhaustive]
+pub struct PlannedRange {
+    /// Byte range in the file.
+    pub range: Range<u64>,
+    /// Row group index in the file.
+    pub row_group: usize,
+    /// Leaf column index in the file.
+    pub column: usize,
+    /// What the range contains.
+    pub kind: PageKind,
+    /// First planned row that this range serves, counted from the start of
+    /// the plan. Used only to order the plan.
+    pub(crate) first_row: u64,
+    /// One past the last planned row that this range serves.
+    pub(crate) last_row: u64,
+    /// The decoding stage that first reads this range.
+    pub(crate) stage: ScanStage,
+    /// `true` if a predicate result can make this range unnecessary.
+    pub(crate) conditional: bool,
+}
+
+impl PlannedRange {
+    /// Number of bytes in this range.
+    pub fn len(&self) -> u64 {
+        self.range.end - self.range.start
+    }
+
+    /// Returns `true` if this range contains no bytes.
+    pub fn is_empty(&self) -> bool {
+        self.range.is_empty()
+    }
+}
+
+/// What a [`PlannedRange`] contains.
+#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
+#[non_exhaustive]
+pub enum PageKind {
+    /// The dictionary page of a column chunk.
+    Dictionary,
+    /// One data page.
+    Data,
+    /// A complete column chunk. The plan uses this when the column has no
+    /// offset index, so page locations are unknown.
+    ColumnChunk,
+}
+
+/// The decoding stage that first reads a [`PlannedRange`].
+///
+/// The decoder reads a column once per row group. A column that a predicate
+/// reads is not read again for the output.
+#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
+pub(crate) enum ScanStage {
+    /// Evaluation of the predicate at this index in the
+    /// [`RowFilter`](crate::arrow::arrow_reader::RowFilter).
+    Predicate(usize),
+    /// Decoding of the output columns.
+    Projection,
+}
+
+/// The byte ranges that a push decoder may still read, in the order that
+/// decoding needs them.
+///
+/// Created by [`ParquetPushDecoder::scan_plan`]. Use it to fetch data before
+/// [`DecodeResult::NeedsData`] asks for it. `NeedsData` stays the exact
+/// request.
+///
+/// * The plan contains every range that the decoder can request after
+///   the plan is made. It can contain more, for example ranges that a
+///   [`RowFilter`] makes unnecessary.
+/// * The plan comes from the current state of the decoder. To plan the
+///   whole scan, call [`ParquetPushDecoder::scan_plan`] once after
+///   [`ParquetPushDecoderBuilder::build`] and keep the iterator. After
+///   [`ParquetPushDecoder::into_builder`], plan again.

Review Comment:
   what does `plan again` mean here? Does this mean that if you rebuild the 
decoder, the previous scan plan may be invalidated and you should call 
`scan_plan` again? If so, perhaps we can just say that explicityl



##########
parquet/src/arrow/push_decoder/scan_plan/mod.rs:
##########
@@ -0,0 +1,1778 @@
+// Licensed to the Apache Software Foundation (ASF) under one
+// or more contributor license agreements.  See the NOTICE file
+// distributed with this work for additional information
+// regarding copyright ownership.  The ASF licenses this file
+// to you under the Apache License, Version 2.0 (the
+// "License"); you may not use this file except in compliance
+// with the License.  You may obtain a copy of the License at
+//
+//   http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing,
+// software distributed under the License is distributed on an
+// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+// KIND, either express or implied.  See the License for the
+// specific language governing permissions and limitations
+// under the License.
+
+//! [`ScanPlan`]: the byte ranges that a push decoder may still read.
+
+use std::cmp::Reverse;
+use std::collections::BinaryHeap;
+use std::iter::FusedIterator;
+use std::ops::Range;
+use std::sync::Arc;
+
+use crate::arrow::ProjectionMask;
+use crate::arrow::arrow_reader::{ReadPlanBuilder, RowSelection};
+use crate::errors::ParquetError;
+use crate::file::metadata::page_index::PageIndexProvider;
+use crate::file::page_index::offset_index::PageLocation;
+
+mod budget;
+mod frontier;
+
+pub(crate) use budget::{BudgetedReadPlan, RowBudget};
+pub(crate) use frontier::{NextRowGroup, RowGroupFrontier};
+
+/// One item of a [`ScanPlan`].
+#[derive(Debug, Clone, PartialEq, Eq)]
+#[non_exhaustive]
+pub struct PlannedRange {
+    /// Byte range in the file.
+    pub range: Range<u64>,
+    /// Row group index in the file.
+    pub row_group: usize,
+    /// Leaf column index in the file.
+    pub column: usize,
+    /// What the range contains.
+    pub kind: PageKind,
+    /// First planned row that this range serves, counted from the start of
+    /// the plan. Used only to order the plan.
+    pub(crate) first_row: u64,
+    /// One past the last planned row that this range serves.
+    pub(crate) last_row: u64,
+    /// The decoding stage that first reads this range.
+    pub(crate) stage: ScanStage,
+    /// `true` if a predicate result can make this range unnecessary.
+    pub(crate) conditional: bool,
+}
+
+impl PlannedRange {
+    /// Number of bytes in this range.
+    pub fn len(&self) -> u64 {
+        self.range.end - self.range.start
+    }
+
+    /// Returns `true` if this range contains no bytes.
+    pub fn is_empty(&self) -> bool {
+        self.range.is_empty()
+    }
+}
+
+/// What a [`PlannedRange`] contains.
+#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
+#[non_exhaustive]
+pub enum PageKind {
+    /// The dictionary page of a column chunk.
+    Dictionary,
+    /// One data page.
+    Data,
+    /// A complete column chunk. The plan uses this when the column has no
+    /// offset index, so page locations are unknown.
+    ColumnChunk,
+}
+
+/// The decoding stage that first reads a [`PlannedRange`].
+///
+/// The decoder reads a column once per row group. A column that a predicate
+/// reads is not read again for the output.
+#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
+pub(crate) enum ScanStage {
+    /// Evaluation of the predicate at this index in the
+    /// [`RowFilter`](crate::arrow::arrow_reader::RowFilter).
+    Predicate(usize),
+    /// Decoding of the output columns.
+    Projection,
+}
+
+/// The byte ranges that a push decoder may still read, in the order that
+/// decoding needs them.
+///
+/// Created by [`ParquetPushDecoder::scan_plan`]. Use it to fetch data before
+/// [`DecodeResult::NeedsData`] asks for it. `NeedsData` stays the exact

Review Comment:
   Maybe we could:
   1. Link to `NeedsData` 
   2. maybe change this to be more readable like "A ScanPlan is an estimate of 
the ranges that may be needed, whereas the returned `NeedsData` from 
`try_decode` are actually needed



##########
parquet/src/arrow/push_decoder/scan_plan/mod.rs:
##########
@@ -0,0 +1,1778 @@
+// Licensed to the Apache Software Foundation (ASF) under one
+// or more contributor license agreements.  See the NOTICE file
+// distributed with this work for additional information
+// regarding copyright ownership.  The ASF licenses this file
+// to you under the Apache License, Version 2.0 (the
+// "License"); you may not use this file except in compliance
+// with the License.  You may obtain a copy of the License at
+//
+//   http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing,
+// software distributed under the License is distributed on an
+// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+// KIND, either express or implied.  See the License for the
+// specific language governing permissions and limitations
+// under the License.
+
+//! [`ScanPlan`]: the byte ranges that a push decoder may still read.
+
+use std::cmp::Reverse;
+use std::collections::BinaryHeap;
+use std::iter::FusedIterator;
+use std::ops::Range;
+use std::sync::Arc;
+
+use crate::arrow::ProjectionMask;
+use crate::arrow::arrow_reader::{ReadPlanBuilder, RowSelection};
+use crate::errors::ParquetError;
+use crate::file::metadata::page_index::PageIndexProvider;
+use crate::file::page_index::offset_index::PageLocation;
+
+mod budget;
+mod frontier;
+
+pub(crate) use budget::{BudgetedReadPlan, RowBudget};
+pub(crate) use frontier::{NextRowGroup, RowGroupFrontier};
+
+/// One item of a [`ScanPlan`].
+#[derive(Debug, Clone, PartialEq, Eq)]
+#[non_exhaustive]
+pub struct PlannedRange {
+    /// Byte range in the file.
+    pub range: Range<u64>,
+    /// Row group index in the file.
+    pub row_group: usize,
+    /// Leaf column index in the file.
+    pub column: usize,
+    /// What the range contains.
+    pub kind: PageKind,
+    /// First planned row that this range serves, counted from the start of
+    /// the plan. Used only to order the plan.
+    pub(crate) first_row: u64,
+    /// One past the last planned row that this range serves.
+    pub(crate) last_row: u64,
+    /// The decoding stage that first reads this range.
+    pub(crate) stage: ScanStage,
+    /// `true` if a predicate result can make this range unnecessary.
+    pub(crate) conditional: bool,
+}
+
+impl PlannedRange {
+    /// Number of bytes in this range.
+    pub fn len(&self) -> u64 {
+        self.range.end - self.range.start
+    }
+
+    /// Returns `true` if this range contains no bytes.
+    pub fn is_empty(&self) -> bool {
+        self.range.is_empty()
+    }
+}
+
+/// What a [`PlannedRange`] contains.
+#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
+#[non_exhaustive]
+pub enum PageKind {
+    /// The dictionary page of a column chunk.
+    Dictionary,
+    /// One data page.
+    Data,
+    /// A complete column chunk. The plan uses this when the column has no
+    /// offset index, so page locations are unknown.
+    ColumnChunk,
+}
+
+/// The decoding stage that first reads a [`PlannedRange`].
+///
+/// The decoder reads a column once per row group. A column that a predicate
+/// reads is not read again for the output.
+#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
+pub(crate) enum ScanStage {
+    /// Evaluation of the predicate at this index in the
+    /// [`RowFilter`](crate::arrow::arrow_reader::RowFilter).
+    Predicate(usize),
+    /// Decoding of the output columns.
+    Projection,
+}
+
+/// The byte ranges that a push decoder may still read, in the order that
+/// decoding needs them.
+///
+/// Created by [`ParquetPushDecoder::scan_plan`]. Use it to fetch data before
+/// [`DecodeResult::NeedsData`] asks for it. `NeedsData` stays the exact
+/// request.
+///
+/// * The plan contains every range that the decoder can request after
+///   the plan is made. It can contain more, for example ranges that a
+///   [`RowFilter`] makes unnecessary.
+/// * The plan comes from the current state of the decoder. To plan the
+///   whole scan, call [`ParquetPushDecoder::scan_plan`] once after
+///   [`ParquetPushDecoderBuilder::build`] and keep the iterator. After
+///   [`ParquetPushDecoder::into_builder`], plan again.
+/// * Row groups are in read order. In a row group, the columns of the
+///   [`RowFilter`] predicates come first, then the other output columns. The
+///   ranges are ordered by the first row that they serve, and the ranges of
+///   one column chunk stay in file order.
+/// * There is one range for each page if the column has an offset index,
+///   and one range for the column chunk if not.
+/// * The plan does not change decoding, does no I/O and does not depend
+///   on pushed data. It is lazy, and its state does not grow with the number
+///   of pages.
+/// * If a row group is not valid, the plan ends before it. The decoder
+///   reports the error.
+///
+/// Keep the fetched ranges in your own cache, and push exactly the ranges

Review Comment:
   this seems like a comment that should go on `push_ranges` (you could link it 
here if you want)



##########
parquet/src/arrow/push_decoder/scan_plan/mod.rs:
##########
@@ -0,0 +1,1778 @@
+// Licensed to the Apache Software Foundation (ASF) under one
+// or more contributor license agreements.  See the NOTICE file
+// distributed with this work for additional information
+// regarding copyright ownership.  The ASF licenses this file
+// to you under the Apache License, Version 2.0 (the
+// "License"); you may not use this file except in compliance
+// with the License.  You may obtain a copy of the License at
+//
+//   http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing,
+// software distributed under the License is distributed on an
+// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+// KIND, either express or implied.  See the License for the
+// specific language governing permissions and limitations
+// under the License.
+
+//! [`ScanPlan`]: the byte ranges that a push decoder may still read.
+
+use std::cmp::Reverse;
+use std::collections::BinaryHeap;
+use std::iter::FusedIterator;
+use std::ops::Range;
+use std::sync::Arc;
+
+use crate::arrow::ProjectionMask;
+use crate::arrow::arrow_reader::{ReadPlanBuilder, RowSelection};
+use crate::errors::ParquetError;
+use crate::file::metadata::page_index::PageIndexProvider;
+use crate::file::page_index::offset_index::PageLocation;
+
+mod budget;
+mod frontier;
+
+pub(crate) use budget::{BudgetedReadPlan, RowBudget};
+pub(crate) use frontier::{NextRowGroup, RowGroupFrontier};
+
+/// One item of a [`ScanPlan`].
+#[derive(Debug, Clone, PartialEq, Eq)]
+#[non_exhaustive]
+pub struct PlannedRange {
+    /// Byte range in the file.
+    pub range: Range<u64>,
+    /// Row group index in the file.
+    pub row_group: usize,
+    /// Leaf column index in the file.
+    pub column: usize,
+    /// What the range contains.
+    pub kind: PageKind,
+    /// First planned row that this range serves, counted from the start of
+    /// the plan. Used only to order the plan.
+    pub(crate) first_row: u64,
+    /// One past the last planned row that this range serves.
+    pub(crate) last_row: u64,
+    /// The decoding stage that first reads this range.
+    pub(crate) stage: ScanStage,
+    /// `true` if a predicate result can make this range unnecessary.
+    pub(crate) conditional: bool,
+}
+
+impl PlannedRange {
+    /// Number of bytes in this range.
+    pub fn len(&self) -> u64 {
+        self.range.end - self.range.start
+    }
+
+    /// Returns `true` if this range contains no bytes.
+    pub fn is_empty(&self) -> bool {
+        self.range.is_empty()
+    }
+}
+
+/// What a [`PlannedRange`] contains.
+#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
+#[non_exhaustive]
+pub enum PageKind {
+    /// The dictionary page of a column chunk.
+    Dictionary,
+    /// One data page.
+    Data,
+    /// A complete column chunk. The plan uses this when the column has no
+    /// offset index, so page locations are unknown.
+    ColumnChunk,
+}
+
+/// The decoding stage that first reads a [`PlannedRange`].
+///
+/// The decoder reads a column once per row group. A column that a predicate
+/// reads is not read again for the output.
+#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
+pub(crate) enum ScanStage {
+    /// Evaluation of the predicate at this index in the
+    /// [`RowFilter`](crate::arrow::arrow_reader::RowFilter).
+    Predicate(usize),
+    /// Decoding of the output columns.
+    Projection,
+}
+
+/// The byte ranges that a push decoder may still read, in the order that
+/// decoding needs them.
+///
+/// Created by [`ParquetPushDecoder::scan_plan`]. Use it to fetch data before
+/// [`DecodeResult::NeedsData`] asks for it. `NeedsData` stays the exact
+/// request.
+///
+/// * The plan contains every range that the decoder can request after
+///   the plan is made. It can contain more, for example ranges that a
+///   [`RowFilter`] makes unnecessary.
+/// * The plan comes from the current state of the decoder. To plan the
+///   whole scan, call [`ParquetPushDecoder::scan_plan`] once after
+///   [`ParquetPushDecoderBuilder::build`] and keep the iterator. After
+///   [`ParquetPushDecoder::into_builder`], plan again.
+/// * Row groups are in read order. In a row group, the columns of the
+///   [`RowFilter`] predicates come first, then the other output columns. The
+///   ranges are ordered by the first row that they serve, and the ranges of
+///   one column chunk stay in file order.
+/// * There is one range for each page if the column has an offset index,
+///   and one range for the column chunk if not.

Review Comment:
   ```suggestion
   ///   and one range for the entire column chunk if not.
   ```



##########
parquet/src/arrow/push_decoder/scan_plan/mod.rs:
##########
@@ -0,0 +1,1778 @@
+// Licensed to the Apache Software Foundation (ASF) under one
+// or more contributor license agreements.  See the NOTICE file
+// distributed with this work for additional information
+// regarding copyright ownership.  The ASF licenses this file
+// to you under the Apache License, Version 2.0 (the
+// "License"); you may not use this file except in compliance
+// with the License.  You may obtain a copy of the License at
+//
+//   http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing,
+// software distributed under the License is distributed on an
+// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+// KIND, either express or implied.  See the License for the
+// specific language governing permissions and limitations
+// under the License.
+
+//! [`ScanPlan`]: the byte ranges that a push decoder may still read.
+
+use std::cmp::Reverse;
+use std::collections::BinaryHeap;
+use std::iter::FusedIterator;
+use std::ops::Range;
+use std::sync::Arc;
+
+use crate::arrow::ProjectionMask;
+use crate::arrow::arrow_reader::{ReadPlanBuilder, RowSelection};
+use crate::errors::ParquetError;
+use crate::file::metadata::page_index::PageIndexProvider;
+use crate::file::page_index::offset_index::PageLocation;
+
+mod budget;
+mod frontier;
+
+pub(crate) use budget::{BudgetedReadPlan, RowBudget};
+pub(crate) use frontier::{NextRowGroup, RowGroupFrontier};
+
+/// One item of a [`ScanPlan`].
+#[derive(Debug, Clone, PartialEq, Eq)]
+#[non_exhaustive]
+pub struct PlannedRange {
+    /// Byte range in the file.
+    pub range: Range<u64>,
+    /// Row group index in the file.
+    pub row_group: usize,
+    /// Leaf column index in the file.
+    pub column: usize,
+    /// What the range contains.
+    pub kind: PageKind,
+    /// First planned row that this range serves, counted from the start of
+    /// the plan. Used only to order the plan.
+    pub(crate) first_row: u64,
+    /// One past the last planned row that this range serves.
+    pub(crate) last_row: u64,
+    /// The decoding stage that first reads this range.
+    pub(crate) stage: ScanStage,
+    /// `true` if a predicate result can make this range unnecessary.
+    pub(crate) conditional: bool,
+}
+
+impl PlannedRange {
+    /// Number of bytes in this range.
+    pub fn len(&self) -> u64 {
+        self.range.end - self.range.start
+    }
+
+    /// Returns `true` if this range contains no bytes.
+    pub fn is_empty(&self) -> bool {
+        self.range.is_empty()
+    }
+}
+
+/// What a [`PlannedRange`] contains.
+#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
+#[non_exhaustive]
+pub enum PageKind {
+    /// The dictionary page of a column chunk.
+    Dictionary,
+    /// One data page.
+    Data,
+    /// A complete column chunk. The plan uses this when the column has no
+    /// offset index, so page locations are unknown.
+    ColumnChunk,
+}
+
+/// The decoding stage that first reads a [`PlannedRange`].
+///
+/// The decoder reads a column once per row group. A column that a predicate
+/// reads is not read again for the output.
+#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
+pub(crate) enum ScanStage {
+    /// Evaluation of the predicate at this index in the
+    /// [`RowFilter`](crate::arrow::arrow_reader::RowFilter).
+    Predicate(usize),
+    /// Decoding of the output columns.
+    Projection,
+}
+
+/// The byte ranges that a push decoder may still read, in the order that
+/// decoding needs them.
+///
+/// Created by [`ParquetPushDecoder::scan_plan`]. Use it to fetch data before
+/// [`DecodeResult::NeedsData`] asks for it. `NeedsData` stays the exact
+/// request.
+///
+/// * The plan contains every range that the decoder can request after
+///   the plan is made. It can contain more, for example ranges that a
+///   [`RowFilter`] makes unnecessary.
+/// * The plan comes from the current state of the decoder. To plan the
+///   whole scan, call [`ParquetPushDecoder::scan_plan`] once after
+///   [`ParquetPushDecoderBuilder::build`] and keep the iterator. After
+///   [`ParquetPushDecoder::into_builder`], plan again.
+/// * Row groups are in read order. In a row group, the columns of the
+///   [`RowFilter`] predicates come first, then the other output columns. The
+///   ranges are ordered by the first row that they serve, and the ranges of
+///   one column chunk stay in file order.
+/// * There is one range for each page if the column has an offset index,
+///   and one range for the column chunk if not.
+/// * The plan does not change decoding, does no I/O and does not depend
+///   on pushed data. It is lazy, and its state does not grow with the number
+///   of pages.
+/// * If a row group is not valid, the plan ends before it. The decoder
+///   reports the error.
+///
+/// Keep the fetched ranges in your own cache, and push exactly the ranges
+/// that `NeedsData` requests. The decoder does not use a requested range
+/// that is split over two pushed buffers, and it releases a pushed buffer
+/// only if its range is equal to a requested range.
+///
+/// # Example
+///
+/// ```
+/// # use std::collections::BTreeMap;
+/// # use std::ops::Range;
+/// # use bytes::Bytes;
+/// # use arrow_array::record_batch;
+/// # use parquet::DecodeResult;
+/// # use parquet::arrow::ArrowWriter;
+/// # use parquet::arrow::arrow_reader::{ArrowReaderMetadata, 
ArrowReaderOptions};
+/// # use parquet::arrow::push_decoder::ParquetPushDecoderBuilder;
+/// # use parquet::file::metadata::PageIndexPolicy;
+/// # use parquet::file::properties::WriterProperties;
+/// # let file = {
+/// #   let mut buffer = vec![];
+/// #   let batch = record_batch!(("a", Int32, [1, 2, 3, 4])).unwrap();
+/// #   let props = 
WriterProperties::builder().set_max_row_group_row_count(Some(2)).build();
+/// #   let mut writer = ArrowWriter::try_new(&mut buffer, batch.schema(), 
Some(props)).unwrap();
+/// #   writer.write(&batch).unwrap();
+/// #   writer.close().unwrap();
+/// #   Bytes::from(buffer)
+/// # };
+/// # let fetch = |range: &Range<u64>| file.slice(range.start as 
usize..range.end as usize);
+/// # // Join cached ranges that cover `range`, or fetch it.
+/// # let read = |cache: &BTreeMap<u64, Bytes>, range: &Range<u64>| {
+/// #     let mut data = Vec::new();
+/// #     let mut position = range.start;
+/// #     while position < range.end {
+/// #         let Some((start, bytes)) = cache.range(..=position).next_back() 
else {
+/// #             return fetch(range);
+/// #         };
+/// #         let end = (start + bytes.len() as u64).min(range.end);
+/// #         if end <= position {
+/// #             return fetch(range);
+/// #         }
+/// #         data.extend_from_slice(&bytes[(position - start) as usize..(end 
- start) as usize]);
+/// #         position = end;
+/// #     }
+/// #     Bytes::from(data)
+/// # };
+/// # let options = 
ArrowReaderOptions::new().with_page_index_policy(PageIndexPolicy::Optional);
+/// # let metadata = ArrowReaderMetadata::load(&file, options).unwrap();
+/// let mut decoder = ParquetPushDecoderBuilder::new_with_metadata(metadata)
+///     .build()
+///     .unwrap();
+///
+/// // Read ahead up to 1 MB into a cache
+/// let mut cache = BTreeMap::new();
+/// let mut cached = 0;
+/// for planned in decoder.scan_plan() {
+///     if cached + planned.len() > 1024 * 1024 {
+///         break;
+///     }
+///     cached += planned.len();
+///     cache.insert(planned.range.start, fetch(&planned.range));
+/// }
+///
+/// // Answer each request from the cache
+/// loop {
+///     match decoder.try_decode().unwrap() {
+///         DecodeResult::NeedsData(ranges) => {
+///             let data = ranges.iter().map(|range| read(&cache, 
range)).collect();

Review Comment:
   It might help here to have a fallback / comment if data wasn't in the cache
   maybe like
   
   ```rust
   ///   DecodeResult::NeedsData(ranges) => {
   ///             // In a real system if the data isn't cached, you would 
fetch it here
   ///             let data = ranges.iter().map(|range| read(&cache, 
range)).collect();
   ```



##########
parquet/src/arrow/push_decoder/scan_plan/mod.rs:
##########
@@ -0,0 +1,1778 @@
+// Licensed to the Apache Software Foundation (ASF) under one
+// or more contributor license agreements.  See the NOTICE file
+// distributed with this work for additional information
+// regarding copyright ownership.  The ASF licenses this file
+// to you under the Apache License, Version 2.0 (the
+// "License"); you may not use this file except in compliance
+// with the License.  You may obtain a copy of the License at
+//
+//   http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing,
+// software distributed under the License is distributed on an
+// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+// KIND, either express or implied.  See the License for the
+// specific language governing permissions and limitations
+// under the License.
+
+//! [`ScanPlan`]: the byte ranges that a push decoder may still read.
+
+use std::cmp::Reverse;
+use std::collections::BinaryHeap;
+use std::iter::FusedIterator;
+use std::ops::Range;
+use std::sync::Arc;
+
+use crate::arrow::ProjectionMask;
+use crate::arrow::arrow_reader::{ReadPlanBuilder, RowSelection};
+use crate::errors::ParquetError;
+use crate::file::metadata::page_index::PageIndexProvider;
+use crate::file::page_index::offset_index::PageLocation;
+
+mod budget;
+mod frontier;
+
+pub(crate) use budget::{BudgetedReadPlan, RowBudget};
+pub(crate) use frontier::{NextRowGroup, RowGroupFrontier};
+
+/// One item of a [`ScanPlan`].
+#[derive(Debug, Clone, PartialEq, Eq)]
+#[non_exhaustive]
+pub struct PlannedRange {
+    /// Byte range in the file.
+    pub range: Range<u64>,
+    /// Row group index in the file.
+    pub row_group: usize,
+    /// Leaf column index in the file.
+    pub column: usize,
+    /// What the range contains.
+    pub kind: PageKind,
+    /// First planned row that this range serves, counted from the start of
+    /// the plan. Used only to order the plan.
+    pub(crate) first_row: u64,
+    /// One past the last planned row that this range serves.
+    pub(crate) last_row: u64,
+    /// The decoding stage that first reads this range.
+    pub(crate) stage: ScanStage,
+    /// `true` if a predicate result can make this range unnecessary.
+    pub(crate) conditional: bool,
+}
+
+impl PlannedRange {
+    /// Number of bytes in this range.
+    pub fn len(&self) -> u64 {
+        self.range.end - self.range.start
+    }
+
+    /// Returns `true` if this range contains no bytes.
+    pub fn is_empty(&self) -> bool {
+        self.range.is_empty()
+    }
+}
+
+/// What a [`PlannedRange`] contains.
+#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
+#[non_exhaustive]
+pub enum PageKind {
+    /// The dictionary page of a column chunk.
+    Dictionary,
+    /// One data page.
+    Data,
+    /// A complete column chunk. The plan uses this when the column has no
+    /// offset index, so page locations are unknown.
+    ColumnChunk,
+}
+
+/// The decoding stage that first reads a [`PlannedRange`].
+///
+/// The decoder reads a column once per row group. A column that a predicate
+/// reads is not read again for the output.
+#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
+pub(crate) enum ScanStage {
+    /// Evaluation of the predicate at this index in the
+    /// [`RowFilter`](crate::arrow::arrow_reader::RowFilter).
+    Predicate(usize),
+    /// Decoding of the output columns.
+    Projection,
+}
+
+/// The byte ranges that a push decoder may still read, in the order that
+/// decoding needs them.
+///
+/// Created by [`ParquetPushDecoder::scan_plan`]. Use it to fetch data before
+/// [`DecodeResult::NeedsData`] asks for it. `NeedsData` stays the exact
+/// request.
+///
+/// * The plan contains every range that the decoder can request after
+///   the plan is made. It can contain more, for example ranges that a
+///   [`RowFilter`] makes unnecessary.
+/// * The plan comes from the current state of the decoder. To plan the
+///   whole scan, call [`ParquetPushDecoder::scan_plan`] once after
+///   [`ParquetPushDecoderBuilder::build`] and keep the iterator. After
+///   [`ParquetPushDecoder::into_builder`], plan again.
+/// * Row groups are in read order. In a row group, the columns of the
+///   [`RowFilter`] predicates come first, then the other output columns. The
+///   ranges are ordered by the first row that they serve, and the ranges of
+///   one column chunk stay in file order.
+/// * There is one range for each page if the column has an offset index,
+///   and one range for the column chunk if not.
+/// * The plan does not change decoding, does no I/O and does not depend
+///   on pushed data. It is lazy, and its state does not grow with the number
+///   of pages.
+/// * If a row group is not valid, the plan ends before it. The decoder

Review Comment:
   What does this mean? How could a row group be invalid? Is this trying to say 
that if the decoder errors not all of the data from the plan may be needed? 
That seems somewhat obvious to me, but it is probably ok to leave



##########
parquet/src/arrow/push_decoder/scan_plan/mod.rs:
##########
@@ -0,0 +1,1778 @@
+// Licensed to the Apache Software Foundation (ASF) under one
+// or more contributor license agreements.  See the NOTICE file
+// distributed with this work for additional information
+// regarding copyright ownership.  The ASF licenses this file
+// to you under the Apache License, Version 2.0 (the
+// "License"); you may not use this file except in compliance
+// with the License.  You may obtain a copy of the License at
+//
+//   http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing,
+// software distributed under the License is distributed on an
+// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+// KIND, either express or implied.  See the License for the
+// specific language governing permissions and limitations
+// under the License.
+
+//! [`ScanPlan`]: the byte ranges that a push decoder may still read.
+
+use std::cmp::Reverse;
+use std::collections::BinaryHeap;
+use std::iter::FusedIterator;
+use std::ops::Range;
+use std::sync::Arc;
+
+use crate::arrow::ProjectionMask;
+use crate::arrow::arrow_reader::{ReadPlanBuilder, RowSelection};
+use crate::errors::ParquetError;
+use crate::file::metadata::page_index::PageIndexProvider;
+use crate::file::page_index::offset_index::PageLocation;
+
+mod budget;
+mod frontier;
+
+pub(crate) use budget::{BudgetedReadPlan, RowBudget};
+pub(crate) use frontier::{NextRowGroup, RowGroupFrontier};
+
+/// One item of a [`ScanPlan`].
+#[derive(Debug, Clone, PartialEq, Eq)]
+#[non_exhaustive]
+pub struct PlannedRange {
+    /// Byte range in the file.
+    pub range: Range<u64>,
+    /// Row group index in the file.
+    pub row_group: usize,
+    /// Leaf column index in the file.
+    pub column: usize,
+    /// What the range contains.
+    pub kind: PageKind,
+    /// First planned row that this range serves, counted from the start of
+    /// the plan. Used only to order the plan.
+    pub(crate) first_row: u64,
+    /// One past the last planned row that this range serves.
+    pub(crate) last_row: u64,
+    /// The decoding stage that first reads this range.
+    pub(crate) stage: ScanStage,
+    /// `true` if a predicate result can make this range unnecessary.
+    pub(crate) conditional: bool,
+}
+
+impl PlannedRange {
+    /// Number of bytes in this range.
+    pub fn len(&self) -> u64 {
+        self.range.end - self.range.start
+    }
+
+    /// Returns `true` if this range contains no bytes.
+    pub fn is_empty(&self) -> bool {
+        self.range.is_empty()
+    }
+}
+
+/// What a [`PlannedRange`] contains.
+#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
+#[non_exhaustive]
+pub enum PageKind {
+    /// The dictionary page of a column chunk.
+    Dictionary,
+    /// One data page.
+    Data,
+    /// A complete column chunk. The plan uses this when the column has no
+    /// offset index, so page locations are unknown.
+    ColumnChunk,
+}
+
+/// The decoding stage that first reads a [`PlannedRange`].
+///
+/// The decoder reads a column once per row group. A column that a predicate
+/// reads is not read again for the output.
+#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
+pub(crate) enum ScanStage {
+    /// Evaluation of the predicate at this index in the
+    /// [`RowFilter`](crate::arrow::arrow_reader::RowFilter).
+    Predicate(usize),
+    /// Decoding of the output columns.
+    Projection,
+}
+
+/// The byte ranges that a push decoder may still read, in the order that
+/// decoding needs them.
+///
+/// Created by [`ParquetPushDecoder::scan_plan`]. Use it to fetch data before
+/// [`DecodeResult::NeedsData`] asks for it. `NeedsData` stays the exact
+/// request.
+///
+/// * The plan contains every range that the decoder can request after
+///   the plan is made. It can contain more, for example ranges that a
+///   [`RowFilter`] makes unnecessary.
+/// * The plan comes from the current state of the decoder. To plan the
+///   whole scan, call [`ParquetPushDecoder::scan_plan`] once after
+///   [`ParquetPushDecoderBuilder::build`] and keep the iterator. After
+///   [`ParquetPushDecoder::into_builder`], plan again.
+/// * Row groups are in read order. In a row group, the columns of the
+///   [`RowFilter`] predicates come first, then the other output columns. The
+///   ranges are ordered by the first row that they serve, and the ranges of
+///   one column chunk stay in file order.
+/// * There is one range for each page if the column has an offset index,
+///   and one range for the column chunk if not.
+/// * The plan does not change decoding, does no I/O and does not depend
+///   on pushed data. It is lazy, and its state does not grow with the number
+///   of pages.
+/// * If a row group is not valid, the plan ends before it. The decoder
+///   reports the error.
+///
+/// Keep the fetched ranges in your own cache, and push exactly the ranges
+/// that `NeedsData` requests. The decoder does not use a requested range
+/// that is split over two pushed buffers, and it releases a pushed buffer
+/// only if its range is equal to a requested range.
+///
+/// # Example
+///
+/// ```
+/// # use std::collections::BTreeMap;
+/// # use std::ops::Range;
+/// # use bytes::Bytes;
+/// # use arrow_array::record_batch;
+/// # use parquet::DecodeResult;
+/// # use parquet::arrow::ArrowWriter;
+/// # use parquet::arrow::arrow_reader::{ArrowReaderMetadata, 
ArrowReaderOptions};
+/// # use parquet::arrow::push_decoder::ParquetPushDecoderBuilder;
+/// # use parquet::file::metadata::PageIndexPolicy;
+/// # use parquet::file::properties::WriterProperties;
+/// # let file = {
+/// #   let mut buffer = vec![];
+/// #   let batch = record_batch!(("a", Int32, [1, 2, 3, 4])).unwrap();
+/// #   let props = 
WriterProperties::builder().set_max_row_group_row_count(Some(2)).build();
+/// #   let mut writer = ArrowWriter::try_new(&mut buffer, batch.schema(), 
Some(props)).unwrap();
+/// #   writer.write(&batch).unwrap();
+/// #   writer.close().unwrap();
+/// #   Bytes::from(buffer)
+/// # };
+/// # let fetch = |range: &Range<u64>| file.slice(range.start as 
usize..range.end as usize);
+/// # // Join cached ranges that cover `range`, or fetch it.
+/// # let read = |cache: &BTreeMap<u64, Bytes>, range: &Range<u64>| {
+/// #     let mut data = Vec::new();
+/// #     let mut position = range.start;
+/// #     while position < range.end {
+/// #         let Some((start, bytes)) = cache.range(..=position).next_back() 
else {
+/// #             return fetch(range);
+/// #         };
+/// #         let end = (start + bytes.len() as u64).min(range.end);
+/// #         if end <= position {
+/// #             return fetch(range);
+/// #         }
+/// #         data.extend_from_slice(&bytes[(position - start) as usize..(end 
- start) as usize]);
+/// #         position = end;
+/// #     }
+/// #     Bytes::from(data)
+/// # };
+/// # let options = 
ArrowReaderOptions::new().with_page_index_policy(PageIndexPolicy::Optional);
+/// # let metadata = ArrowReaderMetadata::load(&file, options).unwrap();
+/// let mut decoder = ParquetPushDecoderBuilder::new_with_metadata(metadata)
+///     .build()
+///     .unwrap();
+///
+/// // Read ahead up to 1 MB into a cache
+/// let mut cache = BTreeMap::new();
+/// let mut cached = 0;
+/// for planned in decoder.scan_plan() {
+///     if cached + planned.len() > 1024 * 1024 {
+///         break;
+///     }
+///     cached += planned.len();
+///     cache.insert(planned.range.start, fetch(&planned.range));
+/// }
+///
+/// // Answer each request from the cache
+/// loop {
+///     match decoder.try_decode().unwrap() {
+///         DecodeResult::NeedsData(ranges) => {
+///             let data = ranges.iter().map(|range| read(&cache, 
range)).collect();
+///             decoder.push_ranges(ranges, data).unwrap();
+///         }
+///         DecodeResult::Data(batch) => println!("{} rows", batch.num_rows()),
+///         DecodeResult::Finished => break,
+///     }
+/// }
+/// ```
+///
+/// [`RowFilter`]: crate::arrow::arrow_reader::RowFilter
+/// [`DecodeResult::NeedsData`]: crate::DecodeResult::NeedsData
+/// [`ParquetPushDecoder::scan_plan`]: super::ParquetPushDecoder::scan_plan
+/// [`ParquetPushDecoder::into_builder`]: 
super::ParquetPushDecoder::into_builder
+/// [`ParquetPushDecoderBuilder::build`]: 
super::ParquetPushDecoderBuilder::build
+#[derive(Debug, Clone)]
+pub struct ScanPlan {
+    /// `None` for an empty plan, for example of a finished decoder.
+    planner: Option<Box<Planner>>,
+}
+
+impl ScanPlan {
+    /// A plan with no ranges.
+    pub(crate) fn empty() -> Self {
+        Self { planner: None }
+    }
+}
+
+/// The state of a [`ScanPlan`] that is not empty.
+#[derive(Debug, Clone)]
+struct Planner {
+    /// The decoder's row-group queue, selections and offset/limit budget, as
+    /// they were when the plan was made.
+    frontier: RowGroupFrontier,
+    /// The row group that the decoder was fetching when the plan was made.
+    /// It is planned first.
+    active_row_group: Option<NextRowGroup>,
+    /// The decoder's projection, predicates and batch size.
+    columns: Arc<StageColumns>,
+    /// Whether an output limit is set.
+    has_limit: bool,
+    /// Whether no row group has been planned yet.
+    at_first_row_group: bool,
+    /// Planned rows before the next row group.
+    next_row: u64,
+    /// The row group being planned, if any.
+    current: Option<RowGroupRanges>,
+    /// Whether planning has ended.
+    done: bool,
+}
+
+/// The columns each decoding stage reads, as [`RowGroupReaderBuilder`]
+/// uses them to decide which bytes a row group needs.
+///
+/// [`RowGroupReaderBuilder`]: super::reader_builder::RowGroupReaderBuilder
+#[derive(Debug)]
+struct StageColumns {
+    /// The output batch size, which aligns cached predicate reads.
+    batch_size: usize,
+    /// Columns in the output.
+    projection: ProjectionMask,
+    /// Columns each [`RowFilter`] predicate reads, in evaluation order.
+    ///
+    /// [`RowFilter`]: crate::arrow::arrow_reader::RowFilter
+    predicate_projections: Vec<ProjectionMask>,
+    /// Predicate columns whose decoded values are cached for the output, if
+    /// any. Their selection is expanded to batch boundaries when fetched.
+    cache_projection: Option<ProjectionMask>,
+}
+
+/// Builds a [`ScanPlan`] from a decoder's row-group frontier and the columns
+/// each decoding stage reads.
+#[derive(Debug)]
+pub(crate) struct ScanPlanBuilder {
+    frontier: RowGroupFrontier,
+    active_row_group: Option<NextRowGroup>,
+    columns: StageColumns,
+}
+
+impl ScanPlanBuilder {
+    /// Plan the row groups in `frontier`, decoding `projection` in batches of
+    /// `batch_size` rows, with no predicates.
+    pub(crate) fn new(
+        frontier: RowGroupFrontier,
+        batch_size: usize,
+        projection: ProjectionMask,
+    ) -> Self {
+        Self {
+            frontier,
+            active_row_group: None,
+            columns: StageColumns {
+                batch_size,
+                projection,
+                predicate_projections: vec![],
+                cache_projection: None,
+            },
+        }
+    }
+
+    /// Set the columns each predicate reads, in evaluation order.
+    pub(crate) fn with_predicate_projections(
+        mut self,
+        predicate_projections: Vec<ProjectionMask>,
+    ) -> Self {
+        self.columns.predicate_projections = predicate_projections;
+        self
+    }
+
+    /// Set the predicate columns whose decoded values are cached for the
+    /// output.
+    pub(crate) fn with_cache_projection(
+        mut self,
+        cache_projection: Option<ProjectionMask>,
+    ) -> Self {
+        self.columns.cache_projection = cache_projection;
+        self
+    }
+
+    /// Plan `active_row_group` first: the row group that the decoder is
+    /// fetching, which the frontier has already handed over.
+    pub(crate) fn with_active_row_group(mut self, active_row_group: 
Option<NextRowGroup>) -> Self {
+        self.active_row_group = active_row_group;
+        self
+    }
+
+    pub(crate) fn build(self) -> ScanPlan {
+        let Self {
+            frontier,
+            active_row_group,
+            columns,
+        } = self;
+        let has_limit = frontier.budget.limit().is_some();
+        let planner = Planner {
+            frontier,
+            active_row_group,
+            columns: Arc::new(columns),
+            has_limit,
+            at_first_row_group: true,
+            next_row: 0,
+            current: None,
+            done: false,
+        };
+        ScanPlan {
+            planner: Some(Box::new(planner)),
+        }
+    }
+}
+
+impl Planner {
+    /// Start planning the next row group the decoder will read.
+    ///
+    /// Returns `Ok(false)` when no row group remains. A row group that the
+    /// offset/limit budget removes is planned with no ranges.
+    fn plan_next_row_group(&mut self) -> Result<bool, ParquetError> {

Review Comment:
   this seems like it is duplicating logic that is elsewhere in the decoder and 
will have to be kept in sync 
   
    I see you have equality tests to keep it consistent, but it would be nice 
to avoid it.
   
   I wonder if you have contemplated using the ScanPlan directly in the 
ParquetDecoder itself -- that way `ParquetDecoder::scan_plan` would return 
`self.scan_plan.clone()` and the calculation of the next row group / needed 
ranges would be in the same logic



##########
parquet/src/arrow/push_decoder/scan_plan/mod.rs:
##########
@@ -0,0 +1,1778 @@
+// Licensed to the Apache Software Foundation (ASF) under one
+// or more contributor license agreements.  See the NOTICE file
+// distributed with this work for additional information
+// regarding copyright ownership.  The ASF licenses this file
+// to you under the Apache License, Version 2.0 (the
+// "License"); you may not use this file except in compliance
+// with the License.  You may obtain a copy of the License at
+//
+//   http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing,
+// software distributed under the License is distributed on an
+// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+// KIND, either express or implied.  See the License for the
+// specific language governing permissions and limitations
+// under the License.
+
+//! [`ScanPlan`]: the byte ranges that a push decoder may still read.
+
+use std::cmp::Reverse;
+use std::collections::BinaryHeap;
+use std::iter::FusedIterator;
+use std::ops::Range;
+use std::sync::Arc;
+
+use crate::arrow::ProjectionMask;
+use crate::arrow::arrow_reader::{ReadPlanBuilder, RowSelection};
+use crate::errors::ParquetError;
+use crate::file::metadata::page_index::PageIndexProvider;
+use crate::file::page_index::offset_index::PageLocation;
+
+mod budget;
+mod frontier;
+
+pub(crate) use budget::{BudgetedReadPlan, RowBudget};
+pub(crate) use frontier::{NextRowGroup, RowGroupFrontier};
+
+/// One item of a [`ScanPlan`].
+#[derive(Debug, Clone, PartialEq, Eq)]
+#[non_exhaustive]
+pub struct PlannedRange {
+    /// Byte range in the file.
+    pub range: Range<u64>,
+    /// Row group index in the file.
+    pub row_group: usize,
+    /// Leaf column index in the file.
+    pub column: usize,
+    /// What the range contains.
+    pub kind: PageKind,
+    /// First planned row that this range serves, counted from the start of
+    /// the plan. Used only to order the plan.
+    pub(crate) first_row: u64,
+    /// One past the last planned row that this range serves.
+    pub(crate) last_row: u64,
+    /// The decoding stage that first reads this range.
+    pub(crate) stage: ScanStage,
+    /// `true` if a predicate result can make this range unnecessary.
+    pub(crate) conditional: bool,
+}
+
+impl PlannedRange {
+    /// Number of bytes in this range.
+    pub fn len(&self) -> u64 {
+        self.range.end - self.range.start
+    }
+
+    /// Returns `true` if this range contains no bytes.
+    pub fn is_empty(&self) -> bool {
+        self.range.is_empty()
+    }
+}
+
+/// What a [`PlannedRange`] contains.
+#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
+#[non_exhaustive]
+pub enum PageKind {
+    /// The dictionary page of a column chunk.
+    Dictionary,
+    /// One data page.
+    Data,
+    /// A complete column chunk. The plan uses this when the column has no
+    /// offset index, so page locations are unknown.
+    ColumnChunk,
+}
+
+/// The decoding stage that first reads a [`PlannedRange`].
+///
+/// The decoder reads a column once per row group. A column that a predicate
+/// reads is not read again for the output.
+#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
+pub(crate) enum ScanStage {
+    /// Evaluation of the predicate at this index in the
+    /// [`RowFilter`](crate::arrow::arrow_reader::RowFilter).
+    Predicate(usize),
+    /// Decoding of the output columns.
+    Projection,
+}
+
+/// The byte ranges that a push decoder may still read, in the order that
+/// decoding needs them.
+///
+/// Created by [`ParquetPushDecoder::scan_plan`]. Use it to fetch data before
+/// [`DecodeResult::NeedsData`] asks for it. `NeedsData` stays the exact
+/// request.
+///
+/// * The plan contains every range that the decoder can request after
+///   the plan is made. It can contain more, for example ranges that a
+///   [`RowFilter`] makes unnecessary.
+/// * The plan comes from the current state of the decoder. To plan the
+///   whole scan, call [`ParquetPushDecoder::scan_plan`] once after
+///   [`ParquetPushDecoderBuilder::build`] and keep the iterator. After
+///   [`ParquetPushDecoder::into_builder`], plan again.
+/// * Row groups are in read order. In a row group, the columns of the
+///   [`RowFilter`] predicates come first, then the other output columns. The
+///   ranges are ordered by the first row that they serve, and the ranges of
+///   one column chunk stay in file order.
+/// * There is one range for each page if the column has an offset index,
+///   and one range for the column chunk if not.
+/// * The plan does not change decoding, does no I/O and does not depend
+///   on pushed data. It is lazy, and its state does not grow with the number
+///   of pages.
+/// * If a row group is not valid, the plan ends before it. The decoder
+///   reports the error.
+///
+/// Keep the fetched ranges in your own cache, and push exactly the ranges
+/// that `NeedsData` requests. The decoder does not use a requested range
+/// that is split over two pushed buffers, and it releases a pushed buffer
+/// only if its range is equal to a requested range.
+///
+/// # Example
+///
+/// ```
+/// # use std::collections::BTreeMap;
+/// # use std::ops::Range;
+/// # use bytes::Bytes;
+/// # use arrow_array::record_batch;
+/// # use parquet::DecodeResult;
+/// # use parquet::arrow::ArrowWriter;
+/// # use parquet::arrow::arrow_reader::{ArrowReaderMetadata, 
ArrowReaderOptions};
+/// # use parquet::arrow::push_decoder::ParquetPushDecoderBuilder;
+/// # use parquet::file::metadata::PageIndexPolicy;
+/// # use parquet::file::properties::WriterProperties;
+/// # let file = {
+/// #   let mut buffer = vec![];
+/// #   let batch = record_batch!(("a", Int32, [1, 2, 3, 4])).unwrap();
+/// #   let props = 
WriterProperties::builder().set_max_row_group_row_count(Some(2)).build();
+/// #   let mut writer = ArrowWriter::try_new(&mut buffer, batch.schema(), 
Some(props)).unwrap();
+/// #   writer.write(&batch).unwrap();
+/// #   writer.close().unwrap();
+/// #   Bytes::from(buffer)
+/// # };
+/// # let fetch = |range: &Range<u64>| file.slice(range.start as 
usize..range.end as usize);
+/// # // Join cached ranges that cover `range`, or fetch it.
+/// # let read = |cache: &BTreeMap<u64, Bytes>, range: &Range<u64>| {
+/// #     let mut data = Vec::new();
+/// #     let mut position = range.start;
+/// #     while position < range.end {
+/// #         let Some((start, bytes)) = cache.range(..=position).next_back() 
else {
+/// #             return fetch(range);
+/// #         };
+/// #         let end = (start + bytes.len() as u64).min(range.end);
+/// #         if end <= position {
+/// #             return fetch(range);
+/// #         }
+/// #         data.extend_from_slice(&bytes[(position - start) as usize..(end 
- start) as usize]);
+/// #         position = end;
+/// #     }
+/// #     Bytes::from(data)
+/// # };
+/// # let options = 
ArrowReaderOptions::new().with_page_index_policy(PageIndexPolicy::Optional);
+/// # let metadata = ArrowReaderMetadata::load(&file, options).unwrap();
+/// let mut decoder = ParquetPushDecoderBuilder::new_with_metadata(metadata)
+///     .build()
+///     .unwrap();
+///
+/// // Read ahead up to 1 MB into a cache
+/// let mut cache = BTreeMap::new();
+/// let mut cached = 0;
+/// for planned in decoder.scan_plan() {
+///     if cached + planned.len() > 1024 * 1024 {
+///         break;
+///     }
+///     cached += planned.len();
+///     cache.insert(planned.range.start, fetch(&planned.range));
+/// }
+///
+/// // Answer each request from the cache
+/// loop {
+///     match decoder.try_decode().unwrap() {
+///         DecodeResult::NeedsData(ranges) => {
+///             let data = ranges.iter().map(|range| read(&cache, 
range)).collect();
+///             decoder.push_ranges(ranges, data).unwrap();
+///         }
+///         DecodeResult::Data(batch) => println!("{} rows", batch.num_rows()),
+///         DecodeResult::Finished => break,
+///     }
+/// }
+/// ```
+///
+/// [`RowFilter`]: crate::arrow::arrow_reader::RowFilter
+/// [`DecodeResult::NeedsData`]: crate::DecodeResult::NeedsData
+/// [`ParquetPushDecoder::scan_plan`]: super::ParquetPushDecoder::scan_plan
+/// [`ParquetPushDecoder::into_builder`]: 
super::ParquetPushDecoder::into_builder
+/// [`ParquetPushDecoderBuilder::build`]: 
super::ParquetPushDecoderBuilder::build
+#[derive(Debug, Clone)]
+pub struct ScanPlan {
+    /// `None` for an empty plan, for example of a finished decoder.
+    planner: Option<Box<Planner>>,
+}
+
+impl ScanPlan {
+    /// A plan with no ranges.
+    pub(crate) fn empty() -> Self {
+        Self { planner: None }
+    }
+}
+
+/// The state of a [`ScanPlan`] that is not empty.
+#[derive(Debug, Clone)]
+struct Planner {
+    /// The decoder's row-group queue, selections and offset/limit budget, as
+    /// they were when the plan was made.
+    frontier: RowGroupFrontier,
+    /// The row group that the decoder was fetching when the plan was made.
+    /// It is planned first.
+    active_row_group: Option<NextRowGroup>,
+    /// The decoder's projection, predicates and batch size.
+    columns: Arc<StageColumns>,
+    /// Whether an output limit is set.
+    has_limit: bool,
+    /// Whether no row group has been planned yet.
+    at_first_row_group: bool,
+    /// Planned rows before the next row group.
+    next_row: u64,
+    /// The row group being planned, if any.
+    current: Option<RowGroupRanges>,
+    /// Whether planning has ended.
+    done: bool,
+}
+
+/// The columns each decoding stage reads, as [`RowGroupReaderBuilder`]
+/// uses them to decide which bytes a row group needs.
+///
+/// [`RowGroupReaderBuilder`]: super::reader_builder::RowGroupReaderBuilder
+#[derive(Debug)]
+struct StageColumns {
+    /// The output batch size, which aligns cached predicate reads.
+    batch_size: usize,
+    /// Columns in the output.
+    projection: ProjectionMask,
+    /// Columns each [`RowFilter`] predicate reads, in evaluation order.
+    ///
+    /// [`RowFilter`]: crate::arrow::arrow_reader::RowFilter
+    predicate_projections: Vec<ProjectionMask>,
+    /// Predicate columns whose decoded values are cached for the output, if
+    /// any. Their selection is expanded to batch boundaries when fetched.
+    cache_projection: Option<ProjectionMask>,
+}
+
+/// Builds a [`ScanPlan`] from a decoder's row-group frontier and the columns
+/// each decoding stage reads.
+#[derive(Debug)]
+pub(crate) struct ScanPlanBuilder {
+    frontier: RowGroupFrontier,
+    active_row_group: Option<NextRowGroup>,
+    columns: StageColumns,
+}
+
+impl ScanPlanBuilder {
+    /// Plan the row groups in `frontier`, decoding `projection` in batches of
+    /// `batch_size` rows, with no predicates.
+    pub(crate) fn new(
+        frontier: RowGroupFrontier,
+        batch_size: usize,
+        projection: ProjectionMask,
+    ) -> Self {
+        Self {
+            frontier,
+            active_row_group: None,
+            columns: StageColumns {
+                batch_size,
+                projection,
+                predicate_projections: vec![],
+                cache_projection: None,
+            },
+        }
+    }
+
+    /// Set the columns each predicate reads, in evaluation order.
+    pub(crate) fn with_predicate_projections(
+        mut self,
+        predicate_projections: Vec<ProjectionMask>,
+    ) -> Self {
+        self.columns.predicate_projections = predicate_projections;
+        self
+    }
+
+    /// Set the predicate columns whose decoded values are cached for the
+    /// output.
+    pub(crate) fn with_cache_projection(
+        mut self,
+        cache_projection: Option<ProjectionMask>,
+    ) -> Self {
+        self.columns.cache_projection = cache_projection;
+        self
+    }
+
+    /// Plan `active_row_group` first: the row group that the decoder is
+    /// fetching, which the frontier has already handed over.
+    pub(crate) fn with_active_row_group(mut self, active_row_group: 
Option<NextRowGroup>) -> Self {
+        self.active_row_group = active_row_group;
+        self
+    }
+
+    pub(crate) fn build(self) -> ScanPlan {
+        let Self {
+            frontier,
+            active_row_group,
+            columns,
+        } = self;
+        let has_limit = frontier.budget.limit().is_some();
+        let planner = Planner {
+            frontier,
+            active_row_group,
+            columns: Arc::new(columns),
+            has_limit,
+            at_first_row_group: true,
+            next_row: 0,
+            current: None,
+            done: false,
+        };
+        ScanPlan {
+            planner: Some(Box::new(planner)),
+        }
+    }
+}
+
+impl Planner {
+    /// Start planning the next row group the decoder will read.
+    ///
+    /// Returns `Ok(false)` when no row group remains. A row group that the
+    /// offset/limit budget removes is planned with no ranges.
+    fn plan_next_row_group(&mut self) -> Result<bool, ParquetError> {
+        // Use the decoder's own row-group walk, so row groups that the
+        // selection or the budget removes are skipped here too.
+        let next_row_group = match self.active_row_group.take() {
+            Some(active) => Some(active),
+            None => self.frontier.next_readable_row_group()?,
+        };
+        let Some(NextRowGroup {
+            row_group_idx,
+            row_count,
+            selection,
+            budget,
+        }) = next_row_group
+        else {
+            return Ok(false);
+        };
+
+        let filtered = self.frontier.has_predicates;
+        let selection = if filtered {
+            // Predicates see every selected row. Offset and limit apply to
+            // their output, so they cannot be applied before decoding.
+            selection
+        } else {
+            // Apply offset and limit as the decoder does before it requests
+            // the output columns.
+            let plan_builder =
+                
ReadPlanBuilder::new(self.columns.batch_size).with_selection(selection);
+            let BudgetedReadPlan {
+                plan_builder,
+                rows_after_budget,
+                remaining_budget,
+                ..
+            } = budget.apply_to_plan(plan_builder, row_count);
+            self.frontier
+                .update_budget_after_row_group(remaining_budget);
+            if rows_after_budget == 0 {
+                self.current = None;
+                return Ok(true);
+            }
+            plan_builder.selection().cloned()
+        };
+
+        let rows = SelectedRows::new(selection.as_ref(), row_count);
+        let first_row = self.next_row;
+        self.next_row += rows.selected_before(row_count);
+
+        let metadata = &self.frontier.parquet_metadata;
+        let page_index = metadata
+            .page_index()
+            .filter(|page_index| page_index.has_offset_indexes())
+            .cloned();
+        let row_group = metadata.row_group(row_group_idx);
+
+        // Predicate columns that are cached for the output are fetched with 
the
+        // selection expanded to batch boundaries. See `fetch_ranges`.
+        let expanded_selection = match (&selection, 
&self.columns.cache_projection) {
+            (Some(selection), Some(_)) if filtered => {
+                
Some(selection.expand_to_batch_boundaries(self.columns.batch_size, row_count))
+            }
+            _ => None,
+        };
+
+        let stages = self
+            .columns
+            .predicate_projections
+            .iter()
+            .enumerate()
+            .map(|(idx, mask)| (ScanStage::Predicate(idx), mask))
+            .chain(std::iter::once((
+                ScanStage::Projection,
+                &self.columns.projection,
+            )));
+
+        let mut planned_columns = vec![false; row_group.columns().len()];
+        let mut stage_plans = vec![];
+        for (stage, mask) in stages {
+            let conditional = filtered
+                && (stage != ScanStage::Predicate(0)
+                    || (self.has_limit && !self.at_first_row_group));
+            let mut columns = vec![];
+            for (column_idx, chunk) in row_group.columns().iter().enumerate() {
+                // The decoder reuses a column that an earlier stage read.
+                if !mask.leaf_included(column_idx) || 
planned_columns[column_idx] {
+                    continue;
+                }
+                planned_columns[column_idx] = true;
+                let is_cached = self
+                    .columns
+                    .cache_projection
+                    .as_ref()
+                    .is_some_and(|cache| cache.leaf_included(column_idx));
+                let (chunk_start, chunk_len) = chunk.byte_range();
+                columns.push(StageColumn {
+                    column_idx,
+                    chunk: chunk_start..chunk_start + chunk_len,
+                    expanded: matches!(stage, ScanStage::Predicate(_))
+                        && is_cached
+                        && expanded_selection.is_some(),
+                });
+            }
+            stage_plans.push(StagePlan {
+                stage,
+                conditional,
+                columns,
+            });
+        }
+        self.at_first_row_group = false;
+
+        self.current = Some(RowGroupRanges {
+            row_group: RowGroupContext {
+                row_group_idx,
+                row_count,
+                first_row,
+                rows,
+                selection,
+                expanded_selection,
+                page_index,
+            },
+            stages: stage_plans.into_iter(),
+            stage: None,
+        });
+        Ok(true)
+    }
+}
+
+impl Iterator for ScanPlan {
+    type Item = PlannedRange;
+
+    fn next(&mut self) -> Option<PlannedRange> {
+        self.planner.as_mut()?.next()
+    }
+}
+
+impl Planner {
+    fn next(&mut self) -> Option<PlannedRange> {
+        loop {
+            if let Some(range) = 
self.current.as_mut().and_then(RowGroupRanges::next) {
+                return Some(range);
+            }
+            if self.done {
+                return None;
+            }
+            match self.plan_next_row_group() {
+                Ok(true) => {}
+                // The decoder reports the same error when it reaches this row
+                // group. The plan is advisory, so it ends here.
+                Ok(false) | Err(_) => {
+                    self.current = None;
+                    self.done = true;
+                }
+            }
+        }
+    }
+}
+
+/// Once [`ScanPlan::next`] returns `None`, it always returns `None`.
+impl FusedIterator for ScanPlan {}
+
+/// The columns that one decoding stage of a row group reads first.
+#[derive(Debug, Clone)]
+struct StagePlan {
+    stage: ScanStage,
+    /// See [`PlannedRange::conditional`].
+    conditional: bool,
+    columns: Vec<StageColumn>,
+}
+
+/// A column chunk that a decoding stage reads.
+#[derive(Debug, Clone)]
+struct StageColumn {

Review Comment:
   I can't help by think this looks very similar to the actual decoder's state 
machine



##########
parquet/src/arrow/push_decoder/scan_plan/mod.rs:
##########
@@ -0,0 +1,1778 @@
+// Licensed to the Apache Software Foundation (ASF) under one
+// or more contributor license agreements.  See the NOTICE file
+// distributed with this work for additional information
+// regarding copyright ownership.  The ASF licenses this file
+// to you under the Apache License, Version 2.0 (the
+// "License"); you may not use this file except in compliance
+// with the License.  You may obtain a copy of the License at
+//
+//   http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing,
+// software distributed under the License is distributed on an
+// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+// KIND, either express or implied.  See the License for the
+// specific language governing permissions and limitations
+// under the License.
+
+//! [`ScanPlan`]: the byte ranges that a push decoder may still read.
+
+use std::cmp::Reverse;
+use std::collections::BinaryHeap;
+use std::iter::FusedIterator;
+use std::ops::Range;
+use std::sync::Arc;
+
+use crate::arrow::ProjectionMask;
+use crate::arrow::arrow_reader::{ReadPlanBuilder, RowSelection};
+use crate::errors::ParquetError;
+use crate::file::metadata::page_index::PageIndexProvider;
+use crate::file::page_index::offset_index::PageLocation;
+
+mod budget;
+mod frontier;
+
+pub(crate) use budget::{BudgetedReadPlan, RowBudget};
+pub(crate) use frontier::{NextRowGroup, RowGroupFrontier};
+
+/// One item of a [`ScanPlan`].
+#[derive(Debug, Clone, PartialEq, Eq)]
+#[non_exhaustive]
+pub struct PlannedRange {
+    /// Byte range in the file.
+    pub range: Range<u64>,
+    /// Row group index in the file.
+    pub row_group: usize,
+    /// Leaf column index in the file.
+    pub column: usize,
+    /// What the range contains.
+    pub kind: PageKind,
+    /// First planned row that this range serves, counted from the start of
+    /// the plan. Used only to order the plan.
+    pub(crate) first_row: u64,
+    /// One past the last planned row that this range serves.
+    pub(crate) last_row: u64,
+    /// The decoding stage that first reads this range.
+    pub(crate) stage: ScanStage,
+    /// `true` if a predicate result can make this range unnecessary.
+    pub(crate) conditional: bool,
+}
+
+impl PlannedRange {
+    /// Number of bytes in this range.
+    pub fn len(&self) -> u64 {
+        self.range.end - self.range.start
+    }
+
+    /// Returns `true` if this range contains no bytes.
+    pub fn is_empty(&self) -> bool {
+        self.range.is_empty()
+    }
+}
+
+/// What a [`PlannedRange`] contains.
+#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
+#[non_exhaustive]
+pub enum PageKind {
+    /// The dictionary page of a column chunk.
+    Dictionary,
+    /// One data page.
+    Data,
+    /// A complete column chunk. The plan uses this when the column has no
+    /// offset index, so page locations are unknown.
+    ColumnChunk,
+}
+
+/// The decoding stage that first reads a [`PlannedRange`].
+///
+/// The decoder reads a column once per row group. A column that a predicate
+/// reads is not read again for the output.
+#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
+pub(crate) enum ScanStage {
+    /// Evaluation of the predicate at this index in the
+    /// [`RowFilter`](crate::arrow::arrow_reader::RowFilter).
+    Predicate(usize),
+    /// Decoding of the output columns.
+    Projection,
+}
+
+/// The byte ranges that a push decoder may still read, in the order that
+/// decoding needs them.
+///
+/// Created by [`ParquetPushDecoder::scan_plan`]. Use it to fetch data before
+/// [`DecodeResult::NeedsData`] asks for it. `NeedsData` stays the exact
+/// request.
+///
+/// * The plan contains every range that the decoder can request after
+///   the plan is made. It can contain more, for example ranges that a
+///   [`RowFilter`] makes unnecessary.
+/// * The plan comes from the current state of the decoder. To plan the
+///   whole scan, call [`ParquetPushDecoder::scan_plan`] once after
+///   [`ParquetPushDecoderBuilder::build`] and keep the iterator. After
+///   [`ParquetPushDecoder::into_builder`], plan again.
+/// * Row groups are in read order. In a row group, the columns of the
+///   [`RowFilter`] predicates come first, then the other output columns. The
+///   ranges are ordered by the first row that they serve, and the ranges of
+///   one column chunk stay in file order.
+/// * There is one range for each page if the column has an offset index,
+///   and one range for the column chunk if not.
+/// * The plan does not change decoding, does no I/O and does not depend
+///   on pushed data. It is lazy, and its state does not grow with the number
+///   of pages.
+/// * If a row group is not valid, the plan ends before it. The decoder
+///   reports the error.
+///
+/// Keep the fetched ranges in your own cache, and push exactly the ranges
+/// that `NeedsData` requests. The decoder does not use a requested range
+/// that is split over two pushed buffers, and it releases a pushed buffer
+/// only if its range is equal to a requested range.
+///
+/// # Example
+///
+/// ```
+/// # use std::collections::BTreeMap;
+/// # use std::ops::Range;
+/// # use bytes::Bytes;
+/// # use arrow_array::record_batch;
+/// # use parquet::DecodeResult;
+/// # use parquet::arrow::ArrowWriter;
+/// # use parquet::arrow::arrow_reader::{ArrowReaderMetadata, 
ArrowReaderOptions};
+/// # use parquet::arrow::push_decoder::ParquetPushDecoderBuilder;
+/// # use parquet::file::metadata::PageIndexPolicy;
+/// # use parquet::file::properties::WriterProperties;
+/// # let file = {
+/// #   let mut buffer = vec![];
+/// #   let batch = record_batch!(("a", Int32, [1, 2, 3, 4])).unwrap();
+/// #   let props = 
WriterProperties::builder().set_max_row_group_row_count(Some(2)).build();
+/// #   let mut writer = ArrowWriter::try_new(&mut buffer, batch.schema(), 
Some(props)).unwrap();
+/// #   writer.write(&batch).unwrap();
+/// #   writer.close().unwrap();
+/// #   Bytes::from(buffer)
+/// # };
+/// # let fetch = |range: &Range<u64>| file.slice(range.start as 
usize..range.end as usize);
+/// # // Join cached ranges that cover `range`, or fetch it.
+/// # let read = |cache: &BTreeMap<u64, Bytes>, range: &Range<u64>| {
+/// #     let mut data = Vec::new();
+/// #     let mut position = range.start;
+/// #     while position < range.end {
+/// #         let Some((start, bytes)) = cache.range(..=position).next_back() 
else {
+/// #             return fetch(range);
+/// #         };
+/// #         let end = (start + bytes.len() as u64).min(range.end);
+/// #         if end <= position {
+/// #             return fetch(range);
+/// #         }
+/// #         data.extend_from_slice(&bytes[(position - start) as usize..(end 
- start) as usize]);
+/// #         position = end;
+/// #     }
+/// #     Bytes::from(data)
+/// # };
+/// # let options = 
ArrowReaderOptions::new().with_page_index_policy(PageIndexPolicy::Optional);
+/// # let metadata = ArrowReaderMetadata::load(&file, options).unwrap();
+/// let mut decoder = ParquetPushDecoderBuilder::new_with_metadata(metadata)
+///     .build()
+///     .unwrap();
+///
+/// // Read ahead up to 1 MB into a cache
+/// let mut cache = BTreeMap::new();
+/// let mut cached = 0;
+/// for planned in decoder.scan_plan() {
+///     if cached + planned.len() > 1024 * 1024 {
+///         break;
+///     }
+///     cached += planned.len();
+///     cache.insert(planned.range.start, fetch(&planned.range));
+/// }
+///
+/// // Answer each request from the cache
+/// loop {
+///     match decoder.try_decode().unwrap() {
+///         DecodeResult::NeedsData(ranges) => {
+///             let data = ranges.iter().map(|range| read(&cache, 
range)).collect();
+///             decoder.push_ranges(ranges, data).unwrap();
+///         }
+///         DecodeResult::Data(batch) => println!("{} rows", batch.num_rows()),
+///         DecodeResult::Finished => break,
+///     }
+/// }
+/// ```
+///
+/// [`RowFilter`]: crate::arrow::arrow_reader::RowFilter
+/// [`DecodeResult::NeedsData`]: crate::DecodeResult::NeedsData
+/// [`ParquetPushDecoder::scan_plan`]: super::ParquetPushDecoder::scan_plan
+/// [`ParquetPushDecoder::into_builder`]: 
super::ParquetPushDecoder::into_builder
+/// [`ParquetPushDecoderBuilder::build`]: 
super::ParquetPushDecoderBuilder::build
+#[derive(Debug, Clone)]
+pub struct ScanPlan {
+    /// `None` for an empty plan, for example of a finished decoder.
+    planner: Option<Box<Planner>>,
+}
+
+impl ScanPlan {
+    /// A plan with no ranges.
+    pub(crate) fn empty() -> Self {
+        Self { planner: None }
+    }
+}
+
+/// The state of a [`ScanPlan`] that is not empty.
+#[derive(Debug, Clone)]
+struct Planner {
+    /// The decoder's row-group queue, selections and offset/limit budget, as
+    /// they were when the plan was made.
+    frontier: RowGroupFrontier,
+    /// The row group that the decoder was fetching when the plan was made.
+    /// It is planned first.
+    active_row_group: Option<NextRowGroup>,
+    /// The decoder's projection, predicates and batch size.
+    columns: Arc<StageColumns>,
+    /// Whether an output limit is set.
+    has_limit: bool,
+    /// Whether no row group has been planned yet.
+    at_first_row_group: bool,
+    /// Planned rows before the next row group.
+    next_row: u64,
+    /// The row group being planned, if any.
+    current: Option<RowGroupRanges>,
+    /// Whether planning has ended.
+    done: bool,
+}
+
+/// The columns each decoding stage reads, as [`RowGroupReaderBuilder`]
+/// uses them to decide which bytes a row group needs.
+///
+/// [`RowGroupReaderBuilder`]: super::reader_builder::RowGroupReaderBuilder
+#[derive(Debug)]
+struct StageColumns {
+    /// The output batch size, which aligns cached predicate reads.
+    batch_size: usize,
+    /// Columns in the output.
+    projection: ProjectionMask,
+    /// Columns each [`RowFilter`] predicate reads, in evaluation order.
+    ///
+    /// [`RowFilter`]: crate::arrow::arrow_reader::RowFilter
+    predicate_projections: Vec<ProjectionMask>,
+    /// Predicate columns whose decoded values are cached for the output, if
+    /// any. Their selection is expanded to batch boundaries when fetched.
+    cache_projection: Option<ProjectionMask>,
+}
+
+/// Builds a [`ScanPlan`] from a decoder's row-group frontier and the columns
+/// each decoding stage reads.
+#[derive(Debug)]
+pub(crate) struct ScanPlanBuilder {
+    frontier: RowGroupFrontier,
+    active_row_group: Option<NextRowGroup>,
+    columns: StageColumns,
+}
+
+impl ScanPlanBuilder {
+    /// Plan the row groups in `frontier`, decoding `projection` in batches of
+    /// `batch_size` rows, with no predicates.
+    pub(crate) fn new(
+        frontier: RowGroupFrontier,
+        batch_size: usize,
+        projection: ProjectionMask,
+    ) -> Self {
+        Self {
+            frontier,
+            active_row_group: None,
+            columns: StageColumns {
+                batch_size,
+                projection,
+                predicate_projections: vec![],
+                cache_projection: None,
+            },
+        }
+    }
+
+    /// Set the columns each predicate reads, in evaluation order.
+    pub(crate) fn with_predicate_projections(
+        mut self,
+        predicate_projections: Vec<ProjectionMask>,
+    ) -> Self {
+        self.columns.predicate_projections = predicate_projections;
+        self
+    }
+
+    /// Set the predicate columns whose decoded values are cached for the
+    /// output.
+    pub(crate) fn with_cache_projection(
+        mut self,
+        cache_projection: Option<ProjectionMask>,
+    ) -> Self {
+        self.columns.cache_projection = cache_projection;
+        self
+    }
+
+    /// Plan `active_row_group` first: the row group that the decoder is
+    /// fetching, which the frontier has already handed over.
+    pub(crate) fn with_active_row_group(mut self, active_row_group: 
Option<NextRowGroup>) -> Self {
+        self.active_row_group = active_row_group;
+        self
+    }
+
+    pub(crate) fn build(self) -> ScanPlan {
+        let Self {
+            frontier,
+            active_row_group,
+            columns,
+        } = self;
+        let has_limit = frontier.budget.limit().is_some();
+        let planner = Planner {
+            frontier,
+            active_row_group,
+            columns: Arc::new(columns),
+            has_limit,
+            at_first_row_group: true,
+            next_row: 0,
+            current: None,
+            done: false,
+        };
+        ScanPlan {
+            planner: Some(Box::new(planner)),
+        }
+    }
+}
+
+impl Planner {
+    /// Start planning the next row group the decoder will read.
+    ///
+    /// Returns `Ok(false)` when no row group remains. A row group that the
+    /// offset/limit budget removes is planned with no ranges.
+    fn plan_next_row_group(&mut self) -> Result<bool, ParquetError> {

Review Comment:
   maybe as a follow on / cleanup PR



##########
parquet/src/arrow/push_decoder/scan_plan/mod.rs:
##########
@@ -0,0 +1,1778 @@
+// Licensed to the Apache Software Foundation (ASF) under one
+// or more contributor license agreements.  See the NOTICE file
+// distributed with this work for additional information
+// regarding copyright ownership.  The ASF licenses this file
+// to you under the Apache License, Version 2.0 (the
+// "License"); you may not use this file except in compliance
+// with the License.  You may obtain a copy of the License at
+//
+//   http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing,
+// software distributed under the License is distributed on an
+// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+// KIND, either express or implied.  See the License for the
+// specific language governing permissions and limitations
+// under the License.
+
+//! [`ScanPlan`]: the byte ranges that a push decoder may still read.
+
+use std::cmp::Reverse;
+use std::collections::BinaryHeap;
+use std::iter::FusedIterator;
+use std::ops::Range;
+use std::sync::Arc;
+
+use crate::arrow::ProjectionMask;
+use crate::arrow::arrow_reader::{ReadPlanBuilder, RowSelection};
+use crate::errors::ParquetError;
+use crate::file::metadata::page_index::PageIndexProvider;
+use crate::file::page_index::offset_index::PageLocation;
+
+mod budget;
+mod frontier;
+
+pub(crate) use budget::{BudgetedReadPlan, RowBudget};
+pub(crate) use frontier::{NextRowGroup, RowGroupFrontier};
+
+/// One item of a [`ScanPlan`].
+#[derive(Debug, Clone, PartialEq, Eq)]
+#[non_exhaustive]
+pub struct PlannedRange {
+    /// Byte range in the file.
+    pub range: Range<u64>,
+    /// Row group index in the file.
+    pub row_group: usize,
+    /// Leaf column index in the file.
+    pub column: usize,
+    /// What the range contains.
+    pub kind: PageKind,
+    /// First planned row that this range serves, counted from the start of
+    /// the plan. Used only to order the plan.
+    pub(crate) first_row: u64,

Review Comment:
   if these are pub crate -- why include them at all? It is probably fine, I am 
just wondering 



##########
parquet/src/arrow/push_decoder/scan_plan/mod.rs:
##########
@@ -0,0 +1,1778 @@
+// Licensed to the Apache Software Foundation (ASF) under one
+// or more contributor license agreements.  See the NOTICE file
+// distributed with this work for additional information
+// regarding copyright ownership.  The ASF licenses this file
+// to you under the Apache License, Version 2.0 (the
+// "License"); you may not use this file except in compliance
+// with the License.  You may obtain a copy of the License at
+//
+//   http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing,
+// software distributed under the License is distributed on an
+// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+// KIND, either express or implied.  See the License for the
+// specific language governing permissions and limitations
+// under the License.
+
+//! [`ScanPlan`]: the byte ranges that a push decoder may still read.
+
+use std::cmp::Reverse;
+use std::collections::BinaryHeap;
+use std::iter::FusedIterator;
+use std::ops::Range;
+use std::sync::Arc;
+
+use crate::arrow::ProjectionMask;
+use crate::arrow::arrow_reader::{ReadPlanBuilder, RowSelection};
+use crate::errors::ParquetError;
+use crate::file::metadata::page_index::PageIndexProvider;
+use crate::file::page_index::offset_index::PageLocation;
+
+mod budget;
+mod frontier;
+
+pub(crate) use budget::{BudgetedReadPlan, RowBudget};
+pub(crate) use frontier::{NextRowGroup, RowGroupFrontier};
+
+/// One item of a [`ScanPlan`].
+#[derive(Debug, Clone, PartialEq, Eq)]
+#[non_exhaustive]
+pub struct PlannedRange {
+    /// Byte range in the file.
+    pub range: Range<u64>,
+    /// Row group index in the file.
+    pub row_group: usize,
+    /// Leaf column index in the file.
+    pub column: usize,
+    /// What the range contains.
+    pub kind: PageKind,
+    /// First planned row that this range serves, counted from the start of
+    /// the plan. Used only to order the plan.
+    pub(crate) first_row: u64,
+    /// One past the last planned row that this range serves.
+    pub(crate) last_row: u64,
+    /// The decoding stage that first reads this range.
+    pub(crate) stage: ScanStage,
+    /// `true` if a predicate result can make this range unnecessary.
+    pub(crate) conditional: bool,
+}
+
+impl PlannedRange {
+    /// Number of bytes in this range.
+    pub fn len(&self) -> u64 {
+        self.range.end - self.range.start
+    }
+
+    /// Returns `true` if this range contains no bytes.
+    pub fn is_empty(&self) -> bool {
+        self.range.is_empty()
+    }
+}
+
+/// What a [`PlannedRange`] contains.
+#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
+#[non_exhaustive]
+pub enum PageKind {
+    /// The dictionary page of a column chunk.
+    Dictionary,
+    /// One data page.
+    Data,
+    /// A complete column chunk. The plan uses this when the column has no
+    /// offset index, so page locations are unknown.
+    ColumnChunk,
+}
+
+/// The decoding stage that first reads a [`PlannedRange`].
+///
+/// The decoder reads a column once per row group. A column that a predicate
+/// reads is not read again for the output.
+#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
+pub(crate) enum ScanStage {
+    /// Evaluation of the predicate at this index in the
+    /// [`RowFilter`](crate::arrow::arrow_reader::RowFilter).
+    Predicate(usize),
+    /// Decoding of the output columns.
+    Projection,
+}
+
+/// The byte ranges that a push decoder may still read, in the order that
+/// decoding needs them.
+///
+/// Created by [`ParquetPushDecoder::scan_plan`]. Use it to fetch data before
+/// [`DecodeResult::NeedsData`] asks for it. `NeedsData` stays the exact
+/// request.
+///
+/// * The plan contains every range that the decoder can request after
+///   the plan is made. It can contain more, for example ranges that a
+///   [`RowFilter`] makes unnecessary.
+/// * The plan comes from the current state of the decoder. To plan the
+///   whole scan, call [`ParquetPushDecoder::scan_plan`] once after
+///   [`ParquetPushDecoderBuilder::build`] and keep the iterator. After
+///   [`ParquetPushDecoder::into_builder`], plan again.
+/// * Row groups are in read order. In a row group, the columns of the
+///   [`RowFilter`] predicates come first, then the other output columns. The
+///   ranges are ordered by the first row that they serve, and the ranges of
+///   one column chunk stay in file order.
+/// * There is one range for each page if the column has an offset index,
+///   and one range for the column chunk if not.
+/// * The plan does not change decoding, does no I/O and does not depend
+///   on pushed data. It is lazy, and its state does not grow with the number
+///   of pages.
+/// * If a row group is not valid, the plan ends before it. The decoder
+///   reports the error.
+///
+/// Keep the fetched ranges in your own cache, and push exactly the ranges
+/// that `NeedsData` requests. The decoder does not use a requested range
+/// that is split over two pushed buffers, and it releases a pushed buffer
+/// only if its range is equal to a requested range.
+///
+/// # Example

Review Comment:
   this is a really nice example



##########
parquet/src/arrow/push_decoder/scan_plan/mod.rs:
##########
@@ -0,0 +1,1778 @@
+// Licensed to the Apache Software Foundation (ASF) under one
+// or more contributor license agreements.  See the NOTICE file
+// distributed with this work for additional information
+// regarding copyright ownership.  The ASF licenses this file
+// to you under the Apache License, Version 2.0 (the
+// "License"); you may not use this file except in compliance
+// with the License.  You may obtain a copy of the License at
+//
+//   http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing,
+// software distributed under the License is distributed on an
+// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+// KIND, either express or implied.  See the License for the
+// specific language governing permissions and limitations
+// under the License.
+
+//! [`ScanPlan`]: the byte ranges that a push decoder may still read.
+
+use std::cmp::Reverse;
+use std::collections::BinaryHeap;
+use std::iter::FusedIterator;
+use std::ops::Range;
+use std::sync::Arc;
+
+use crate::arrow::ProjectionMask;
+use crate::arrow::arrow_reader::{ReadPlanBuilder, RowSelection};
+use crate::errors::ParquetError;
+use crate::file::metadata::page_index::PageIndexProvider;
+use crate::file::page_index::offset_index::PageLocation;
+
+mod budget;
+mod frontier;
+
+pub(crate) use budget::{BudgetedReadPlan, RowBudget};
+pub(crate) use frontier::{NextRowGroup, RowGroupFrontier};
+
+/// One item of a [`ScanPlan`].
+#[derive(Debug, Clone, PartialEq, Eq)]
+#[non_exhaustive]
+pub struct PlannedRange {
+    /// Byte range in the file.
+    pub range: Range<u64>,
+    /// Row group index in the file.
+    pub row_group: usize,
+    /// Leaf column index in the file.
+    pub column: usize,
+    /// What the range contains.
+    pub kind: PageKind,
+    /// First planned row that this range serves, counted from the start of
+    /// the plan. Used only to order the plan.
+    pub(crate) first_row: u64,
+    /// One past the last planned row that this range serves.
+    pub(crate) last_row: u64,
+    /// The decoding stage that first reads this range.
+    pub(crate) stage: ScanStage,
+    /// `true` if a predicate result can make this range unnecessary.
+    pub(crate) conditional: bool,
+}
+
+impl PlannedRange {
+    /// Number of bytes in this range.
+    pub fn len(&self) -> u64 {
+        self.range.end - self.range.start
+    }
+
+    /// Returns `true` if this range contains no bytes.
+    pub fn is_empty(&self) -> bool {
+        self.range.is_empty()
+    }
+}
+
+/// What a [`PlannedRange`] contains.
+#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
+#[non_exhaustive]
+pub enum PageKind {
+    /// The dictionary page of a column chunk.
+    Dictionary,
+    /// One data page.
+    Data,
+    /// A complete column chunk. The plan uses this when the column has no
+    /// offset index, so page locations are unknown.
+    ColumnChunk,
+}
+
+/// The decoding stage that first reads a [`PlannedRange`].
+///
+/// The decoder reads a column once per row group. A column that a predicate
+/// reads is not read again for the output.
+#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
+pub(crate) enum ScanStage {
+    /// Evaluation of the predicate at this index in the
+    /// [`RowFilter`](crate::arrow::arrow_reader::RowFilter).
+    Predicate(usize),
+    /// Decoding of the output columns.
+    Projection,
+}
+
+/// The byte ranges that a push decoder may still read, in the order that
+/// decoding needs them.
+///
+/// Created by [`ParquetPushDecoder::scan_plan`]. Use it to fetch data before
+/// [`DecodeResult::NeedsData`] asks for it. `NeedsData` stays the exact
+/// request.
+///
+/// * The plan contains every range that the decoder can request after

Review Comment:
   I think another important property to mention is that the ranges are 
generated on demand



-- 
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.

To unsubscribe, e-mail: [email protected]

For queries about this service, please contact Infrastructure at:
[email protected]

Reply via email to