yyanyy commented on code in PR #58298: URL: https://github.com/apache/spark/pull/58298#discussion_r4009415724
########## sql/core/src/main/scala/org/apache/spark/sql/execution/datasources/v2/CapturedSchemaProjection.scala: ########## @@ -0,0 +1,310 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.spark.sql.execution.datasources.v2 + +import org.apache.spark.SparkException +import org.apache.spark.sql.catalyst.SQLConfHelper +import org.apache.spark.sql.catalyst.analysis.Resolver +import org.apache.spark.sql.catalyst.expressions.{Alias, ArrayTransform, AttributeReference, CreateNamedStruct, Expression, GetStructField, If, IsNull, KnownNotNull, LambdaFunction, Literal, MetadataAttributeWithLogicalName, NamedLambdaVariable, TaggingExpression, TransformKeys, TransformValues, UnresolvedNamedLambdaVariable} +import org.apache.spark.sql.catalyst.plans.logical.{LogicalPlan, Project} +import org.apache.spark.sql.catalyst.util.MetadataColumnHelper +import org.apache.spark.sql.types.{ArrayType, DataType, MapType, Metadata, StructType} + +/** + * Rebinds a relation that reads a current table schema to output attributes captured from an + * earlier compatible schema. The current schema is exposed by the relation so its output remains + * aligned with the physical scan, while a projection recreates the captured output for the + * already-analyzed parent plan. + */ +private[sql] object CapturedSchemaProjection extends SQLConfHelper { + + /** + * Prevents [[CreateNamedStruct]] from inheriting metadata from a field value while leaving the + * value's type, nullability, evaluation, and code generation unchanged. + */ + private case class MetadataPropagationBarrier(child: Expression) extends TaggingExpression { + override protected def withNewChildInternal( + newChild: Expression): MetadataPropagationBarrier = copy(child = newChild) + } + + def rebindToCapturedSchema(relation: DataSourceV2Relation): LogicalPlan = { + // The relation still carries the output captured at analysis time; only its table has been + // swapped for the current one. + val capturedOutput = relation.output + val resolver = conf.resolver + val current = DataSourceV2Relation.create( + relation.table, + relation.catalog, + relation.identifier, + relation.options, + relation.timeTravelSpec) + val currentMetadataOutput = current.metadataOutput + val currentMetadata = capturedOutput.filter(_.isMetadataCol).map { captured => + val logicalName = metadataLogicalName(captured) + matchName(currentMetadataOutput, logicalName, resolver)(metadataLogicalName) + .map(pos => currentMetadataOutput(pos)) + .getOrElse { + // The connector still reports this metadata column, so it can only be absent here + // because a data column has taken its name and the connector suppresses rather than + // renames the conflict (`canRenameConflictingMetadataColumns`). Validation owns + // rejecting that. + unexpectedSchemaChange( + s"captured metadata column $logicalName is missing from the current relation") + } + } + + val currentOutput = current.output ++ currentMetadata + + // Refresh may visit an already rebound relation. Preserve its attributes so the projection + // above it continues to reference valid expression IDs. + // + // A further schema change on such a relation adds a second projection instead of replacing + // the first. Only the cache stores a refreshed plan, so the effect is limited to that entry: + // it stops matching the single projection a query rebuilds from its own captured output, and + // is no longer reused. Results stay correct. + if (sameOutputShape(capturedOutput, currentOutput)) { + return relation + } + + val capturedIndex = new AttributeIndex(capturedOutput, resolver) + val reboundOutput = currentOutput.map { currentAttr => + capturedIndex.get(currentAttr).filter(canReuse(_, currentAttr)).getOrElse(currentAttr) + } + val reboundRelation = relation.copy(output = reboundOutput) + + val reboundIndex = new AttributeIndex(reboundOutput, resolver) + val projectList = capturedOutput.map { capturedAttr => + val currentAttr = reboundIndex.get(capturedAttr).getOrElse { + unexpectedSchemaChange( + s"captured column ${capturedAttr.name} is missing from current table ${relation.name}") + } + if (currentAttr.exprId == capturedAttr.exprId && + sameAttributeShape(currentAttr, capturedAttr)) { + currentAttr + } else { + if (currentAttr.nullable != capturedAttr.nullable) { + unexpectedSchemaChange( + s"nullability changed for captured column ${capturedAttr.name} in ${relation.name}") + } + val projected = projectToType( + currentAttr, currentAttr.dataType, capturedAttr.dataType, resolver) + if (projected.dataType != capturedAttr.dataType || + projected.nullable != capturedAttr.nullable) { + unexpectedSchemaChange( + s"failed to recreate captured column ${capturedAttr.name} in ${relation.name}") + } + Alias(projected, capturedAttr.name)( + exprId = capturedAttr.exprId, + qualifier = capturedAttr.qualifier, + explicitMetadata = Some(capturedAttr.metadata)) + } + } + + Project(projectList, reboundRelation) + } + + private[v2] def projectToType( + input: Expression, + from: DataType, + to: DataType, + resolver: Resolver): Expression = { + if (from == to) { + return input + } + + val projected = (from, to) match { + case (fromStruct: StructType, toStruct: StructType) => Review Comment: Confirmed, and I've done the test half of this but not the fix. On the fix, I'd still prefer to keep it separate, and to be concrete about why: `GetStructField(If(IsNull(person), ...), 0)` loses pruning because `SchemaPruning.getRootFields` only prunes through `SelectedField`, which requires the `GetStructField` chain to be rooted at an attribute. Rooted at an `If`, it falls through to the child-recursion case, reaches the bare `person` attribute and marks the whole struct as required. The predicate fails to translate for the same reason. So an optimizer-friendly form has to keep the null guard while leaving a shape those rules recognise, and it also has to preserve the captured field names, order, metadata and nullability. `UpdateFields` is the promising direction since `SimplifyExtractValueOps` already unwraps `GetStructField(UpdateFields(...))`, but `WithField` builds a `StructField` with no metadata and doesn't cover reordering, arrays or maps — so this becomes a hybrid `projectToType` plus new rules and tests for both paths, which I thi nk earns its own change. Worth noting the scope: this affects only a plan analyzed before a field was added inside a struct, array or map and executed after. A top-level addition reuses expression IDs and produces a plain attribute project list, and any query analyzed after the change prunes normally. Results are correct throughout; the cost is reading wider than necessary in that window. -- This is an automated message from the Apache Git Service. To respond to the message, please log on to GitHub and use the URL above to go to the specific comment. To unsubscribe, e-mail: [email protected] For queries about this service, please contact Infrastructure at: [email protected] --------------------------------------------------------------------- To unsubscribe, e-mail: [email protected] For additional commands, e-mail: [email protected]
