pitrou commented on code in PR #50945:
URL: https://github.com/apache/arrow/pull/50945#discussion_r3981030682
##########
cpp/src/arrow/json/chunker.cc:
##########
@@ -165,6 +91,78 @@ class ParsingBoundaryFinder : public BoundaryFinder {
int64_t* out_pos, int64_t* num_found) override {
return Status::NotImplemented("ParsingBoundaryFinder::FindNth");
}
+
+ private:
+ simdjson::ondemand::parser parser_;
+ // A persistent buffer to keep padded contents for simdjson.
+ // This should be more efficient than allocating a new padded_string
everytime.
+ std::string buffer_;
+
+ simdjson::padded_string_view GetPaddedStringView(std::string_view partial,
+ std::string_view block =
{}) {
+ // Adjust buffer size without copying old contents.
+ buffer_.clear();
+ buffer_.reserve(partial.size() + block.size() +
simdjson::SIMDJSON_PADDING);
+ buffer_.append(partial);
+ buffer_.append(block);
+ // XXX Hopefully this upholds for all std::string implementations
+ DCHECK_GE(buffer_.capacity() - buffer_.size(), simdjson::SIMDJSON_PADDING);
+ auto view = simdjson::padded_string_view(buffer_);
+ DCHECK(view.has_sufficient_padding());
+ return view;
+ }
+
+ // Consume the first or last JSON object (depending on `until_end`)
+ // and return the consumed JSON byte length, or 0 if no valid document
+ // can be parsed.
+ Result<size_t> ConsumeWholeObject(simdjson::padded_string_view input, bool
until_end) {
+ simdjson::ondemand::document_stream stream;
+ // XXX Should be pass a specific batch_size?
+ // The default value used by simdjson is 1MB, probably enough for most
purposes.
+ RETURN_NOT_OK(ToStatus(parser_.iterate_many(input).get(stream)));
+ auto it = stream.begin();
+ if (it == stream.end()) {
+ // Empty input (only whitespace?)
+ return 0;
+ }
+
+ int64_t consumed_length = 0;
+ if (!until_end) {
+ // Parsing the first document only.
+ if (!ConsumeDocument(it).ok()) {
+ // Could be either a partial document or invalid JSON, we'll let
+ // followup chunker or parser calls decide.
+ return 0;
+ }
+ // current_index() is the start of the current document;
+ // source() is the complete source span of the current document.
+ consumed_length = it.current_index() + it.source().size();
+ } else {
+ while (it != stream.end()) {
+ if (!ConsumeDocument(it).ok()) {
+ // Could be either a partial document or invalid JSON, we'll let
+ // followup chunker or parser calls decide.
+ break;
+ }
+ consumed_length = it.current_index() + it.source().size();
+ ++it;
+ }
+ }
+ if (consumed_length > 0) {
+ // If we found at least one document, also consume its trailing
whitespace
+ // to avoid stray bytes at the end of the stream.
+ consumed_length += ConsumeWhitespace(input.substr(consumed_length));
+ }
Review Comment:
> Or is the contract that `consumed_length` is the same as `input.size()` if
we consume the last element of the input?
I'm not sure it's explicitly spelled out, but that would have my preference.
If I disable this code I get some test errors, though they might be harmless.
--
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.
To unsubscribe, e-mail: [email protected]
For queries about this service, please contact Infrastructure at:
[email protected]