viirya commented on code in PR #56334: URL: https://github.com/apache/spark/pull/56334#discussion_r3564806818
########## sql/core/src/main/scala/org/apache/spark/sql/execution/columnar/ArrowCachedBatchSerializer.scala: ########## @@ -0,0 +1,1588 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.spark.sql.execution.columnar + +import java.io.{ByteArrayInputStream, ByteArrayOutputStream} +import java.nio.channels.Channels + +import scala.jdk.CollectionConverters._ + +import org.apache.arrow.compression.{Lz4CompressionCodec, ZstdCompressionCodec} +import org.apache.arrow.vector.{VectorLoader, VectorSchemaRoot, VectorUnloader} +import org.apache.arrow.vector.compression.{CompressionCodec, NoCompressionCodec} +import org.apache.arrow.vector.ipc.{ReadChannel, WriteChannel} +import org.apache.arrow.vector.ipc.message.{ArrowRecordBatch, MessageSerializer} + +import org.apache.spark.{SparkException, TaskContext} +import org.apache.spark.rdd.RDD +import org.apache.spark.sql.catalyst.InternalRow +import org.apache.spark.sql.catalyst.expressions.Attribute +import org.apache.spark.sql.catalyst.expressions.codegen.UnsafeRowWriter +import org.apache.spark.sql.catalyst.types.DataTypeUtils +import org.apache.spark.sql.columnar.{CachedBatch, SimpleMetricsCachedBatchSerializer} +import org.apache.spark.sql.errors.ExecutionErrors +import org.apache.spark.sql.execution.arrow.ArrowWriter +import org.apache.spark.sql.internal.SQLConf +import org.apache.spark.sql.types._ +import org.apache.spark.sql.util.ArrowUtils +import org.apache.spark.sql.vectorized.{ArrowColumnVector, ColumnarBatch, ColumnVector} +import org.apache.spark.storage.StorageLevel +import org.apache.spark.unsafe.types.{TimestampNanosVal, UTF8String} +import org.apache.spark.util.Utils + +/** + * A [[CachedBatchSerializer]] that uses Apache Arrow as the cache format. + * + * This serializer: + * - Supports both row-based (InternalRow) and columnar (ColumnarBatch) input + * - Stores each batch as an internal, schema-less encapsulated Arrow RecordBatch message with + * optional compression (zstd/lz4); the schema is reconstructed from the relation's attributes + * on read (see [[ArrowCachedBatch]]) + * - Enables zero-copy columnar reads when output is ColumnarBatch + * - Uses off-heap memory via Arrow allocators during encode/decode + * - Collects per-column statistics for partition pruning + * + * Configuration options: + * - spark.sql.cache.serializer: Set to this class name to enable + * - spark.sql.execution.arrow.maxRecordsPerBatch: Max rows per cached batch + * - spark.sql.execution.arrow.maxBytesPerBatch: Max bytes per cached batch + * - spark.sql.execution.arrow.compression.codec: Compression (none/zstd/lz4) + * - spark.sql.execution.arrow.compression.zstd.level: zstd compression level + * - spark.sql.execution.arrow.cache.prefetch.enabled: Enable background prefetch of the next + * batch while the current one is being consumed + * - spark.sql.inMemoryColumnarStorage.enableVectorizedReader: Enable columnar output + */ +class ArrowCachedBatchSerializer extends SimpleMetricsCachedBatchSerializer { + + // supportsColumnarInput selects the columnar-vs-row input path; it does not gate which schemas + // this serializer accepts. The cache framework has no per-type fallback to another serializer + // (whatever spark.sql.cache.serializer selects handles every cached relation), so returning + // false here only routes input through convertInternalRowToCachedBatch, which is still this + // serializer. Type support is enforced once per partition by checkSupportedSchema below; the + // only real precondition for columnar input is that the plan can produce columnar output, which + // InMemoryRelation already checks via cachedPlan.supportsColumnar before calling this. + override def supportsColumnarInput(schema: Seq[Attribute]): Boolean = true + + override def convertInternalRowToCachedBatch( + input: RDD[InternalRow], + schema: Seq[Attribute], + storageLevel: StorageLevel, + conf: SQLConf): RDD[CachedBatch] = { + ArrowCachedBatchSerializer.checkSupportedSchema(schema) + // Capture config values on driver before RDD transformation + val sparkSchema = DataTypeUtils.fromAttributes(schema) + val maxRecordsPerBatch = conf.arrowMaxRecordsPerBatch + val maxBytesPerBatch = conf.arrowMaxBytesPerBatch + val timeZoneId = conf.sessionLocalTimeZone + val compressionCodecName = conf.arrowCompressionCodec + val compressionLevel = conf.arrowZstdCompressionLevel + + input.mapPartitionsInternal { rowIterator => + new InternalRowToArrowCachedBatchIterator( + rowIterator, + schema, + sparkSchema, + maxRecordsPerBatch, + maxBytesPerBatch, + timeZoneId, + compressionCodecName, + compressionLevel) + } + } + + override def convertColumnarBatchToCachedBatch( + input: RDD[ColumnarBatch], + schema: Seq[Attribute], + storageLevel: StorageLevel, + conf: SQLConf): RDD[CachedBatch] = { + ArrowCachedBatchSerializer.checkSupportedSchema(schema) + // Capture config values on driver before RDD transformation + val sparkSchema = DataTypeUtils.fromAttributes(schema) + val timeZoneId = conf.sessionLocalTimeZone + val compressionCodecName = conf.arrowCompressionCodec + val compressionLevel = conf.arrowZstdCompressionLevel + + input.mapPartitionsInternal { batchIterator => + new ColumnarBatchToArrowCachedBatchIterator( + batchIterator, + schema, + sparkSchema, + timeZoneId, + compressionCodecName, + compressionLevel) + } + } + + override def supportsColumnarOutput(schema: StructType): Boolean = { + // Always support columnar output with Arrow + true + } + + override def vectorTypes(attributes: Seq[Attribute], conf: SQLConf): Option[Seq[String]] = { + Option(Seq.fill(attributes.length)(classOf[ArrowColumnVector].getName)) + } + + override def convertCachedBatchToColumnarBatch( + input: RDD[CachedBatch], + cacheAttributes: Seq[Attribute], + selectedAttributes: Seq[Attribute], + conf: SQLConf): RDD[ColumnarBatch] = { + val cacheSchema = DataTypeUtils.fromAttributes(cacheAttributes) + val selectedSchema = DataTypeUtils.fromAttributes(selectedAttributes) + val columnIndices = + selectedAttributes.map(a => cacheAttributes.map(o => o.exprId).indexOf(a.exprId)).toArray + // Capture config on driver + val timeZoneId = conf.sessionLocalTimeZone + val prefetchEnabled = conf.arrowCachePrefetchEnabled + + input.mapPartitionsInternal { batchIterator => + new ArrowCachedBatchToColumnarBatchIterator( + batchIterator, + cacheSchema, + selectedSchema, + columnIndices, + timeZoneId, + prefetchEnabled) + } + } + + override def convertCachedBatchToInternalRow( + input: RDD[CachedBatch], + cacheAttributes: Seq[Attribute], + selectedAttributes: Seq[Attribute], + conf: SQLConf): RDD[InternalRow] = { + val cacheSchema = DataTypeUtils.fromAttributes(cacheAttributes) + val selectedSchema = DataTypeUtils.fromAttributes(selectedAttributes) + val timeZoneId = conf.sessionLocalTimeZone + + // Calculate column indices for projection + val selectedIndices = selectedAttributes.map { attr => + cacheAttributes.indexWhere(_.exprId == attr.exprId) + }.toArray + + // Check if all selected types can use the fast path. + // Types not handled by ArrowColumnReader must use the fallback path. + val needsFallback = selectedSchema.fields.exists { f => + f.dataType match { + case _: ArrayType | _: StructType | _: MapType => true + case CalendarIntervalType | VariantType | NullType => true + case _: UserDefinedType[_] => true + // Geometry/Geography are represented as an Arrow struct (srid + wkb); the fast-path + // ArrowColumnReader does not handle them, so route them through the fallback. + case _: GeometryType | _: GeographyType => true + // Nanosecond timestamps write a 16-byte TimestampNanosVal payload into the UnsafeRow; + // the fast-path typed readers only write fixed primitives, so use the fallback, which + // reads through ArrowColumnVector.getTimestampNTZNanos/getTimestampLTZNanos. + case _: TimestampNTZNanosType | _: TimestampLTZNanosType => true + case _ => false + } + } + + if (needsFallback) { + // Fall back to columnar-to-row conversion via ColumnarBatch for complex types. + // Use UnsafeProjection to convert ColumnarBatchRow to UnsafeRow. + convertCachedBatchToColumnarBatch(input, cacheAttributes, selectedAttributes, conf) + .mapPartitionsInternal { batchIter => + val toUnsafe = org.apache.spark.sql.catalyst.expressions.UnsafeProjection.create( + selectedSchema) + batchIter.flatMap { batch => + val numRows = batch.numRows() + new Iterator[InternalRow] { + private var rowIdx = 0 + override def hasNext: Boolean = rowIdx < numRows + override def next(): InternalRow = { + val row = batch.getRow(rowIdx) + rowIdx += 1 + toUnsafe(row) + } + } + } + } + } else { + val prefetchEnabled = conf.arrowCachePrefetchEnabled + input.mapPartitionsInternal { batchIterator => + new ArrowCachedBatchToInternalRowIterator( + batchIterator, + cacheSchema, + selectedSchema, + selectedIndices, + timeZoneId, + prefetchEnabled) + } + } + } +} + +/** + * Companion object with shared utility methods for Arrow cache serialization. + */ +private object ArrowCachedBatchSerializer { + + /** + * Fail fast, once per partition on the driver-facing entry points, if any column type cannot be + * represented by the Arrow cache. This is the actual capability gate (supportsColumnarInput only + * chooses the input path). Without it, an unsupported type would otherwise surface as a less + * obvious failure deeper in schema conversion or statistics collection. + */ + def checkSupportedSchema(schema: Seq[Attribute]): Unit = { + schema.find(attr => !ArrowUtils.isSupportedByArrow(attr.dataType)).foreach { attr => + // Use the structured user-facing condition (UNSUPPORTED_DATATYPE) rather than an internal + // error: an unsupported column type is a user-visible limitation, and the docs promise this + // condition. It is also the same condition toArrowSchema raises, so callers see one + // condition for the capability regardless of which layer detects it first. + throw ExecutionErrors.unsupportedDataTypeError(attr.dataType) + } + } + + // scalastyle:off caselocale + def createCompressionCodec( + codecName: String, + compressionLevel: Int): CompressionCodec = { + codecName.toLowerCase match { + case "none" => NoCompressionCodec.INSTANCE + // The codec instance must be constructed directly so that compressionLevel is honored: + // CompressionCodec.Factory.createCodec(codecType) ignores the level and builds a codec at + // the default level. The level only matters on the write side; the read side looks up the + // codec by the type recorded in the IPC message. + case "zstd" => new ZstdCompressionCodec(compressionLevel) + case "lz4" => new Lz4CompressionCodec() + case other => + throw SparkException.internalError( + s"Unsupported Arrow compression codec: $other. Supported values: none, zstd, lz4") + } + } + // scalastyle:on caselocale + + def serializeBatch(batch: ArrowRecordBatch): Array[Byte] = { + val out = new ByteArrayOutputStream() + val writeChannel = new WriteChannel(Channels.newChannel(out)) + MessageSerializer.serialize(writeChannel, batch) + out.toByteArray + } + + /** + * Byte offset of the unscaled low-order word within a 16-byte Arrow Decimal128 slot, for the + * given native byte order. Arrow Java writes decimal values in native byte order + * (DecimalUtility.writeLongToArrowBuf / writeBigDecimalToArrowBuf): on little-endian platforms + * the low-order word occupies the first 8 bytes and the sign-extension word follows, while on + * big-endian platforms the order is reversed and the low-order word occupies the last 8 bytes. + * Reading the wrong word on a big-endian JVM turns positive compact decimals into 0 and + * negative ones into an unscaled -1. + */ + def compactDecimalUnscaledOffset(nativeOrder: java.nio.ByteOrder): Long = + if (nativeOrder == java.nio.ByteOrder.LITTLE_ENDIAN) 0L else 8L + + /** + * Shut down a prefetch worker during task cleanup without leaking the root it may have produced. + * + * The prefetch worker deserializes the next batch into a fresh [[VectorSchemaRoot]] off-thread. + * If task completion runs while a result is in flight (e.g. a LIMIT consumer stops early), + * cancelling and discarding the future would drop a root that was already (or is about to be) + * produced, and the subsequent `allocator.close()` would fail with "Memory was leaked by query". + * + * This stops accepting new work, waits for the worker to finish so no root is produced after we + * stop looking, then closes any completed result. Always returns null so the caller can null out + * its future reference. Safe to call with a null executor or future. + */ + def drainAndClosePrefetch( + executor: java.util.concurrent.ExecutorService, + future: java.util.concurrent.Future[VectorSchemaRoot]): java.util.concurrent.Future[ + VectorSchemaRoot] = { + // Drain and join the worker uninterruptibly, then close any root it produced, before the + // caller closes the allocator. This runs from a task-completion listener, which can fire with + // the task thread already interrupted (e.g. a killed task). If we let awaitTermination or + // future.get observe the interrupt and bail early, the worker could still be allocating into, + // or have already returned, a root that we then neither join nor close -- and the subsequent + // allocator.close() would race the worker or fail with "Memory was leaked by query". So we + // defer every interruption -- whether present on entry or delivered while blocked (throwing + // InterruptedException clears the status, so it must be recorded here or it is lost) -- and + // restore the flag once draining is done. + var wasInterrupted = Thread.interrupted() + try { + if (executor != null) { + executor.shutdown() + var terminated = false + while (!terminated) { + try { + terminated = + executor.awaitTermination(Long.MaxValue, java.util.concurrent.TimeUnit.NANOSECONDS) + } catch { + // Record the interruption and keep waiting: we must not leave the worker running. + case _: InterruptedException => wasInterrupted = true + } + } + } + if (future != null) { + try { + // The worker has terminated, so this does not block; close the root it produced. + val root = future.get() + if (root != null) { + root.close() + } + } catch { + // The batch was never produced (cancelled/failed); nothing to close. + case _: java.util.concurrent.CancellationException => + case _: java.util.concurrent.ExecutionException => + case _: InterruptedException => wasInterrupted = true + } + } + } finally { + if (wasInterrupted) { + Thread.currentThread().interrupt() + } + } + null + } + + def createColumnStats(dataType: DataType): ColumnStats = { + dataType match { + case BooleanType => new BooleanColumnStats + case ByteType => new ByteColumnStats + case ShortType => new ShortColumnStats + case IntegerType => new IntColumnStats + case DateType => new IntColumnStats // Date is stored as Int + case LongType => new LongColumnStats + case TimestampType => new LongColumnStats // Timestamp is stored as Long + case TimestampNTZType => new LongColumnStats // TimestampNTZ is stored as Long + // Nanosecond timestamps use the TimestampNanosVal-aware collector (min/max bounds), the + // same one the default cache serializer uses for these types. + case _: TimestampNTZNanosType | _: TimestampLTZNanosType => new TimestampNanosColumnStats + case FloatType => new FloatColumnStats + case DoubleType => new DoubleColumnStats + case st: StringType => new StringColumnStats(st) + case BinaryType => new BinaryColumnStats + case dt: DecimalType => new DecimalColumnStats(dt) + case CalendarIntervalType => new IntervalColumnStats + case _: YearMonthIntervalType => new IntColumnStats // stored as Int + case _: DayTimeIntervalType => new LongColumnStats // stored as Long + case _: TimeType => new LongColumnStats // Time is stored as Long (nanoseconds) + case VariantType => new VariantColumnStats + // Geometry/Geography collect size/count without min/max bounds. Their physical value is a + // BinaryView (not Array[Byte]), so GeoColumnStats reads it via getBinaryView rather than + // BinaryColumnStats' getBinary, which would throw ClassCastException on a row that stores a + // BinaryView. They are also AtomicTypes that ColumnType (used by ObjectColumnStats) does not + // handle, so they must be matched explicitly here. + case _: GeometryType | _: GeographyType => new GeoColumnStats + // Unwrap UDTs to the same collector their underlying type would use. isSupportedByArrow + // accepts a UDT whenever its sqlType is supported (including Variant/Geometry/Geography), + // but ObjectColumnStats -> ColumnType(udt.sqlType) only unwraps one level and has no case + // for those types, so it would throw UNSUPPORTED_DATATYPE during materialization. Recursing + // here keeps the capability check and the statistics path in agreement. + case udt: UserDefinedType[_] => createColumnStats(udt.sqlType) + case _ => new ObjectColumnStats(dataType) + } + } + + def buildStatisticsFromCollectors( + collectors: Array[ColumnStats], + schema: Seq[Attribute]): InternalRow = { + val stats = collectors.flatMap { collector => + val collected = collector.collectedStatistics + // ColumnStats returns: [lowerBound, upperBound, nullCount, count, sizeInBytes] + Seq(collected(0), collected(1), collected(2), collected(3), collected(4)) + } + InternalRow.fromSeq(stats.toSeq) + } + + def collectStatistics( + root: VectorSchemaRoot, + schema: Seq[Attribute]): InternalRow = { + val rowCount = root.getRowCount + val vectors = root.getFieldVectors.asScala.toSeq + + // Collect stats for each column: lowerBound, upperBound, nullCount, rowCount, sizeInBytes + val stats = schema.zip(vectors).flatMap { case (attr, vector) => + val nullCount = (0 until rowCount).count(i => vector.isNull(i)) Review Comment: Applied. I verified the semantic-equivalence claims: `NullVector.getNullCount` returns `valueCount` (all rows null, matching the per-row loop), validity-backed vectors count set bits word-at-a-time, and the struct-backed lossless types count the struct's own validity buffer -- the same buffer `isNull` reads. Since this was the last remaining per-value pass for non-orderable columns on the zero-copy re-cache path, it directly cuts cache-build cost there. -- This is an automated message from the Apache Git Service. To respond to the message, please log on to GitHub and use the URL above to go to the specific comment. To unsubscribe, e-mail: [email protected] For queries about this service, please contact Infrastructure at: [email protected] --------------------------------------------------------------------- To unsubscribe, e-mail: [email protected] For additional commands, e-mail: [email protected]
