This is an automated email from the ASF dual-hosted git repository.

pitrou pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/arrow.git


The following commit(s) were added to refs/heads/main by this push:
     new d8890e0b12 GH-50275: [C++][CSV] avoid int32 overflow in block parser 
value counts (#50074)
d8890e0b12 is described below

commit d8890e0b12269c12a80c4fe3004c02053d83fd4a
Author: metsw24-max <[email protected]>
AuthorDate: Mon Jun 29 21:10:58 2026 +0530

    GH-50275: [C++][CSV] avoid int32 overflow in block parser value counts 
(#50074)
    
    ### Rationale for this change
    
    The CSV block parser sizes its per-chunk value array from `num_cols`, the 
column count inferred from the first line of the input, times the rows-in-chunk 
count. `PresizedValueDescWriter` computes `2 + num_rows * num_cols`, and 
`ParseSpecialized` computes `num_cols_ * (num_rows_ - start) * 10`, both in 
`int32_t`. A CSV whose first line carries a few million fields pushes these 
products past `INT32_MAX`, which is signed-integer-overflow UB (UBSan flags 
both expressions).
    
    ### What changes are included in this PR?
    
    Raise an error if the `num_cols` would prevent the block size from fitting 
in a 32 bits integer.
    
    ### Are these changes tested?
    
    With a new unit test.
    
    ### Are there any user-facing changes?
    
    No.
    
    * GitHub Issue: #50275
    
    Lead-authored-by: metsw24-max <[email protected]>
    Co-authored-by: Sayed Kaif <[email protected]>
    Signed-off-by: Antoine Pitrou <[email protected]>
---
 cpp/src/arrow/csv/parser.cc      | 18 +++++++++++++++---
 cpp/src/arrow/csv/parser_test.cc | 11 +++++++++++
 2 files changed, 26 insertions(+), 3 deletions(-)

diff --git a/cpp/src/arrow/csv/parser.cc b/cpp/src/arrow/csv/parser.cc
index e83855336d..54dc54ae7e 100644
--- a/cpp/src/arrow/csv/parser.cc
+++ b/cpp/src/arrow/csv/parser.cc
@@ -204,7 +204,8 @@ class PresizedValueDescWriter : public 
ValueDescWriter<PresizedValueDescWriter>
   // however we allow for one extraneous write in case of excessive columns,
   // hence `2 + num_rows * num_cols` (see explanation in PushValue below).
   PresizedValueDescWriter(MemoryPool* pool, int32_t num_rows, int32_t num_cols)
-      : ValueDescWriter(pool, /*values_capacity=*/2 + num_rows * num_cols) {}
+      : ValueDescWriter(
+            pool, /*values_capacity=*/2 + static_cast<int64_t>(num_rows) * 
num_cols) {}
 
   void PushValue(ParsedValueDesc v) {
     DCHECK_LT(values_size_, values_capacity_);
@@ -535,8 +536,8 @@ class BlockParserImpl {
       // Use bulk filter only if average value length is >= 10 bytes,
       // as the bulk filter has a fixed cost that isn't compensated
       // when values are too short.
-      const int64_t bulk_filter_threshold =
-          batch_.num_cols_ * (batch_.num_rows_ - start_num_rows) * 10;
+      const int64_t bulk_filter_threshold = 
static_cast<int64_t>(batch_.num_cols_) *
+                                            (batch_.num_rows_ - 
start_num_rows) * 10;
       use_bulk_filter_ = (data - *out_data) > bulk_filter_threshold;
     }
 
@@ -604,6 +605,17 @@ class BlockParserImpl {
           rows_in_chunk = std::min(kTargetChunkSize, max_num_rows_ - 
batch_.num_rows_);
         }
 
+        // The values array holds one ParsedValueDesc per cell and those 
offsets
+        // are 31-bit, so the number of values in a chunk must fit in an int32.
+        // A first line with millions of fields can drive `num_cols_` high 
enough
+        // to overflow that, so error out rather than presize past the limit.
+        if (static_cast<int64_t>(rows_in_chunk) * batch_.num_cols_ >
+            std::numeric_limits<int32_t>::max()) {
+          return Status::Invalid("CSV parser: row group of ", rows_in_chunk, " 
rows x ",
+                                 batch_.num_cols_,
+                                 " columns exceeds the maximum number of 
values");
+        }
+
         ARROW_ASSIGN_OR_RAISE(
             auto values_writer,
             PresizedValueDescWriter::Make(pool_, rows_in_chunk, 
batch_.num_cols_));
diff --git a/cpp/src/arrow/csv/parser_test.cc b/cpp/src/arrow/csv/parser_test.cc
index 0b5e717509..abe57dd2e8 100644
--- a/cpp/src/arrow/csv/parser_test.cc
+++ b/cpp/src/arrow/csv/parser_test.cc
@@ -666,6 +666,17 @@ TEST(BlockParser, MismatchingNumColumns) {
   }
 }
 
+TEST(BlockParser, TooManyValues) {
+  // A first line carrying millions of fields drives num_cols high enough that
+  // the per-chunk value count (rows x columns) would overflow the 31-bit value
+  // offset, so the parser errors out instead of overflowing.
+  uint32_t out_size;
+  BlockParser parser(ParseOptions::Defaults(), /*num_cols=*/5000000);
+  Status st = Parse(parser, MakeCSVData({"a,b\n"}), &out_size);
+  EXPECT_RAISES_WITH_MESSAGE_THAT(
+      Invalid, testing::HasSubstr("exceeds the maximum number of values"), st);
+}
+
 TEST(BlockParser, MismatchingNumColumnsHandler) {
   struct CustomHandler {
     operator InvalidRowHandler() {

Reply via email to