Fokko commented on code in PR #3494:
URL: https://github.com/apache/parquet-java/pull/3494#discussion_r3156907453
##########
parquet-column/src/main/java/org/apache/parquet/column/values/plain/PlainValuesReader.java:
##########
@@ -19,120 +19,92 @@
package org.apache.parquet.column.values.plain;
import java.io.IOException;
+import java.nio.ByteBuffer;
+import java.nio.ByteOrder;
import org.apache.parquet.bytes.ByteBufferInputStream;
import org.apache.parquet.bytes.LittleEndianDataInputStream;
import org.apache.parquet.column.values.ValuesReader;
-import org.apache.parquet.io.ParquetDecodingException;
import org.slf4j.Logger;
import org.slf4j.LoggerFactory;
/**
- * Plain encoding for float, double, int, long
+ * Plain encoding for float, double, int, long.
+ *
+ * <p>Reads directly from a {@link ByteBuffer} with {@link
ByteOrder#LITTLE_ENDIAN} byte order,
+ * bypassing the {@link LittleEndianDataInputStream} wrapper to avoid
per-value virtual dispatch
+ * overhead. The underlying page data is obtained as a single contiguous
{@link ByteBuffer} via
+ * {@link ByteBufferInputStream#slice(int)}.
*/
public abstract class PlainValuesReader extends ValuesReader {
private static final Logger LOG =
LoggerFactory.getLogger(PlainValuesReader.class);
- protected LittleEndianDataInputStream in;
+ ByteBuffer buffer;
@Override
public void initFromPage(int valueCount, ByteBufferInputStream stream)
throws IOException {
LOG.debug("init from page at offset {} for length {}", stream.position(),
stream.available());
- this.in = new LittleEndianDataInputStream(stream.remainingStream());
+ int available = stream.available();
+ if (available > 0) {
+ this.buffer = stream.slice(available).order(ByteOrder.LITTLE_ENDIAN);
+ } else {
+ this.buffer = ByteBuffer.allocate(0).order(ByteOrder.LITTLE_ENDIAN);
+ }
}
@Override
public void skip() {
skip(1);
}
- void skipBytesFully(int n) throws IOException {
- int skipped = 0;
- while (skipped < n) {
- skipped += in.skipBytes(n - skipped);
- }
- }
-
public static class DoublePlainValuesReader extends PlainValuesReader {
@Override
public void skip(int n) {
- try {
- skipBytesFully(n * 8);
- } catch (IOException e) {
- throw new ParquetDecodingException("could not skip " + n + " double
values", e);
- }
+ buffer.position(buffer.position() + n * 8);
Review Comment:
When skipping, should we validate bounds?
--
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.
To unsubscribe, e-mail: [email protected]
For queries about this service, please contact Infrastructure at:
[email protected]
---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]