This is an automated email from the ASF dual-hosted git repository.
pitrou pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/arrow.git
The following commit(s) were added to refs/heads/main by this push:
new 65974f5316 GH-51470: [C++] Delimit JSON documents without parsing them
(#51472)
65974f5316 is described below
commit 65974f5316ab9404605e7f5a2fa6c890e9c42888
Author: Alexander Taepper <[email protected]>
AuthorDate: Mon Sep 28 12:21:44 2026 +0200
GH-51470: [C++] Delimit JSON documents without parsing them (#51472)
### Rationale for this change
Resolves #51470:
`simdjson` has a multi-stage parser that first finds structural elements in
the entire buffer (including document boundaries). Only the second stage, which
is lazy, actually need to parse the documents.
Similarily `parse_many` already computes all document boundaries and we can
iterate its output to consume the documents without parsing any documents:
https://github.com/simdjson/simdjson/blob/6913ee1a17a57d23a2e9f458d8e04e96cd43ecab/doc/parse_many.md?plain=1#L85-L88
### What changes are included in this PR?
This delimits JSON documents without fully parsing them
### Are these changes tested?
Yes.
before:
```
ChunkJSONPrettyPrinted 528656
ns 528396 ns 1340
bytes_per_second=394.823Mi/s json_size=218.757k
ChunkJSONPrettyPrintedMultipleBlocks 664758
ns 625186 ns 1143 block_size=27.344k
bytes_per_second=333.697Mi/s json_size=218.757k
```
after:
```
ChunkJSONPrettyPrinted 91815
ns 91811 ns 7358
bytes_per_second=2.21906Gi/s json_size=218.757k
ChunkJSONPrettyPrintedMultipleBlocks 197358
ns 167654 ns 3719 block_size=27.344k
bytes_per_second=1.2152Gi/s json_size=218.757k
```
### Are there any user-facing changes?
No.
### Was AI used for this PR?
In accordance to the [AI generation
guidelines](https://arrow.apache.org/docs/dev/developers/overview.html#ai-generated-code),
please disclose below whether and how AI was used in this PR.
**PR code and description written by:**
- [X] Human
- [ ] (AI assisted in code drafting for the initial version)
**Reviewed before submission by:**
- [X] Human
- [ ] AI
- [ ] Not reviewed
* GitHub Issue: #51470
Authored-by: Alexander Taepper <[email protected]>
Signed-off-by: Antoine Pitrou <[email protected]>
---
cpp/src/arrow/json/chunker.cc | 21 ++++++---------------
cpp/src/arrow/json/reader_test.cc | 9 +--------
2 files changed, 7 insertions(+), 23 deletions(-)
diff --git a/cpp/src/arrow/json/chunker.cc b/cpp/src/arrow/json/chunker.cc
index f69c45b938..4bff06e2ca 100644
--- a/cpp/src/arrow/json/chunker.cc
+++ b/cpp/src/arrow/json/chunker.cc
@@ -42,17 +42,8 @@ int64_t ConsumeWhitespace(std::string_view view) {
return static_cast<int64_t>(ws_count);
}
-Status ConsumeDocument(simdjson::ondemand::document_stream::iterator& it) {
- ARROW_ASSIGN_OR_RAISE(
- auto document, internal::ResolveSimdjsonResult(*it, "Failed to get JSON
document"));
- ARROW_ASSIGN_OR_RAISE(
- auto value,
- internal::ResolveSimdjsonResult(document.get_value(), "Failed to get
JSON value"));
- return internal::ConsumeJsonValue(value);
-}
-
// A BoundaryFinder implementation that assumes JSON objects can contain raw
newlines,
-// and uses actual JSON parsing to delimit them.
+// and uses the structural indexes computed by simdjson to delimit them.
class ParsingBoundaryFinder : public BoundaryFinder {
public:
explicit ParsingBoundaryFinder(MemoryPool* pool) : pool_(pool) {}
@@ -122,8 +113,8 @@ class ParsingBoundaryFinder : public BoundaryFinder {
}
// Find the first or last JSON object (depending on `find_last`)
- // and return the consumed JSON byte length, or 0 if no valid document
- // can be parsed.
+ // and return the consumed JSON byte length, or 0 if no complete document
+ // can be found.
Result<size_t> FindDocument(simdjson::padded_string_view input, bool
find_last) {
simdjson::ondemand::document_stream stream;
// XXX Should be pass a specific batch_size?
@@ -137,8 +128,8 @@ class ParsingBoundaryFinder : public BoundaryFinder {
int64_t consumed_length = 0;
if (!find_last) {
- // Parsing the first document only.
- if (!ConsumeDocument(it).ok()) {
+ // Delimiting the first document only.
+ if (it.error()) {
// Could be either a partial document or invalid JSON, we'll let
// followup chunker or parser calls decide.
return 0;
@@ -148,7 +139,7 @@ class ParsingBoundaryFinder : public BoundaryFinder {
consumed_length = it.current_index() + it.source().size();
} else {
while (it != stream.end()) {
- if (!ConsumeDocument(it).ok()) {
+ if (it.error()) {
// Could be either a partial document or invalid JSON, we'll let
// followup chunker or parser calls decide.
break;
diff --git a/cpp/src/arrow/json/reader_test.cc
b/cpp/src/arrow/json/reader_test.cc
index 7dcd30a0eb..549a494805 100644
--- a/cpp/src/arrow/json/reader_test.cc
+++ b/cpp/src/arrow/json/reader_test.cc
@@ -743,16 +743,9 @@ TEST_P(StreamingReaderTest,
PropagateErrorsNonLinewiseChunker) {
read_options_.block_size = 10;
parse_options_.newlines_in_values = true;
- ASSERT_OK_AND_ASSIGN(reader, MakeReader(bad_first_block));
- AssertReadNext(reader, &batch);
- EXPECT_EQ(reader->bytes_processed(), 7);
- ASSERT_BATCHES_EQUAL(*RecordBatchFromJSON(test_schema, "[{\"i\":0}]"),
*batch);
-
EXPECT_RAISES_WITH_MESSAGE_THAT(Invalid,
::testing::StartsWith("Invalid: JSON parse
error"),
- reader->ReadNext(&batch));
- EXPECT_EQ(reader->bytes_processed(), 7);
- AssertReadEnd(reader);
+ MakeReader(bad_first_block));
ASSERT_OK_AND_ASSIGN(reader, MakeReader(bad_middle_blocks));
AssertReadNext(reader, &batch);