This is an automated email from the ASF dual-hosted git repository.

pitrou pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/arrow.git


The following commit(s) were added to refs/heads/main by this push:
     new 65974f5316 GH-51470: [C++] Delimit JSON documents without parsing them 
(#51472)
65974f5316 is described below

commit 65974f5316ab9404605e7f5a2fa6c890e9c42888
Author: Alexander Taepper <[email protected]>
AuthorDate: Mon Sep 28 12:21:44 2026 +0200

    GH-51470: [C++] Delimit JSON documents without parsing them (#51472)
    
    ### Rationale for this change
    
    Resolves #51470:
    
    `simdjson` has a multi-stage parser that first finds structural elements in 
the entire buffer (including document boundaries). Only the second stage, which 
is lazy, actually need to parse the documents.
    
    Similarily `parse_many` already computes all document boundaries and we can 
iterate its output to consume the documents without parsing any documents:
    
https://github.com/simdjson/simdjson/blob/6913ee1a17a57d23a2e9f458d8e04e96cd43ecab/doc/parse_many.md?plain=1#L85-L88
    
    ### What changes are included in this PR?
    
    This delimits JSON documents without fully parsing them
    
    ### Are these changes tested?
    
    Yes.
    
    before:
    
    ```
    ChunkJSONPrettyPrinted                                             528656 
ns       528396 ns         1340
    bytes_per_second=394.823Mi/s json_size=218.757k
    ChunkJSONPrettyPrintedMultipleBlocks                               664758 
ns       625186 ns         1143 block_size=27.344k
    bytes_per_second=333.697Mi/s json_size=218.757k
    ```
    
    after:
    
    ```
    ChunkJSONPrettyPrinted                                              91815 
ns        91811 ns         7358
    bytes_per_second=2.21906Gi/s json_size=218.757k
    ChunkJSONPrettyPrintedMultipleBlocks                               197358 
ns       167654 ns         3719 block_size=27.344k
    bytes_per_second=1.2152Gi/s json_size=218.757k
    ```
    
    ### Are there any user-facing changes?
    
    No.
    
    ### Was AI used for this PR?
    
    In accordance to the [AI generation 
guidelines](https://arrow.apache.org/docs/dev/developers/overview.html#ai-generated-code),
 please disclose below whether and how AI was used in this PR.
    
    **PR code and description written by:**
    
    - [X] Human
    - [ ] (AI assisted in code drafting for the initial version)
    
    **Reviewed before submission by:**
    
    - [X] Human
    - [ ] AI
    - [ ] Not reviewed
    
    * GitHub Issue: #51470
    
    Authored-by: Alexander Taepper <[email protected]>
    Signed-off-by: Antoine Pitrou <[email protected]>
---
 cpp/src/arrow/json/chunker.cc     | 21 ++++++---------------
 cpp/src/arrow/json/reader_test.cc |  9 +--------
 2 files changed, 7 insertions(+), 23 deletions(-)

diff --git a/cpp/src/arrow/json/chunker.cc b/cpp/src/arrow/json/chunker.cc
index f69c45b938..4bff06e2ca 100644
--- a/cpp/src/arrow/json/chunker.cc
+++ b/cpp/src/arrow/json/chunker.cc
@@ -42,17 +42,8 @@ int64_t ConsumeWhitespace(std::string_view view) {
   return static_cast<int64_t>(ws_count);
 }
 
-Status ConsumeDocument(simdjson::ondemand::document_stream::iterator& it) {
-  ARROW_ASSIGN_OR_RAISE(
-      auto document, internal::ResolveSimdjsonResult(*it, "Failed to get JSON 
document"));
-  ARROW_ASSIGN_OR_RAISE(
-      auto value,
-      internal::ResolveSimdjsonResult(document.get_value(), "Failed to get 
JSON value"));
-  return internal::ConsumeJsonValue(value);
-}
-
 // A BoundaryFinder implementation that assumes JSON objects can contain raw 
newlines,
-// and uses actual JSON parsing to delimit them.
+// and uses the structural indexes computed by simdjson to delimit them.
 class ParsingBoundaryFinder : public BoundaryFinder {
  public:
   explicit ParsingBoundaryFinder(MemoryPool* pool) : pool_(pool) {}
@@ -122,8 +113,8 @@ class ParsingBoundaryFinder : public BoundaryFinder {
   }
 
   // Find the first or last JSON object (depending on `find_last`)
-  // and return the consumed JSON byte length, or 0 if no valid document
-  // can be parsed.
+  // and return the consumed JSON byte length, or 0 if no complete document
+  // can be found.
   Result<size_t> FindDocument(simdjson::padded_string_view input, bool 
find_last) {
     simdjson::ondemand::document_stream stream;
     // XXX Should be pass a specific batch_size?
@@ -137,8 +128,8 @@ class ParsingBoundaryFinder : public BoundaryFinder {
 
     int64_t consumed_length = 0;
     if (!find_last) {
-      // Parsing the first document only.
-      if (!ConsumeDocument(it).ok()) {
+      // Delimiting the first document only.
+      if (it.error()) {
         // Could be either a partial document or invalid JSON, we'll let
         // followup chunker or parser calls decide.
         return 0;
@@ -148,7 +139,7 @@ class ParsingBoundaryFinder : public BoundaryFinder {
       consumed_length = it.current_index() + it.source().size();
     } else {
       while (it != stream.end()) {
-        if (!ConsumeDocument(it).ok()) {
+        if (it.error()) {
           // Could be either a partial document or invalid JSON, we'll let
           // followup chunker or parser calls decide.
           break;
diff --git a/cpp/src/arrow/json/reader_test.cc 
b/cpp/src/arrow/json/reader_test.cc
index 7dcd30a0eb..549a494805 100644
--- a/cpp/src/arrow/json/reader_test.cc
+++ b/cpp/src/arrow/json/reader_test.cc
@@ -743,16 +743,9 @@ TEST_P(StreamingReaderTest, 
PropagateErrorsNonLinewiseChunker) {
   read_options_.block_size = 10;
   parse_options_.newlines_in_values = true;
 
-  ASSERT_OK_AND_ASSIGN(reader, MakeReader(bad_first_block));
-  AssertReadNext(reader, &batch);
-  EXPECT_EQ(reader->bytes_processed(), 7);
-  ASSERT_BATCHES_EQUAL(*RecordBatchFromJSON(test_schema, "[{\"i\":0}]"), 
*batch);
-
   EXPECT_RAISES_WITH_MESSAGE_THAT(Invalid,
                                   ::testing::StartsWith("Invalid: JSON parse 
error"),
-                                  reader->ReadNext(&batch));
-  EXPECT_EQ(reader->bytes_processed(), 7);
-  AssertReadEnd(reader);
+                                  MakeReader(bad_first_block));
 
   ASSERT_OK_AND_ASSIGN(reader, MakeReader(bad_middle_blocks));
   AssertReadNext(reader, &batch);

Reply via email to