This is an automated email from the ASF dual-hosted git repository.
Abacn pushed a commit to branch master
in repository https://gitbox.apache.org/repos/asf/beam.git
The following commit(s) were added to refs/heads/master by this push:
new e671af6c779 [Spark][#36841] Register Spark's streaming internals with
Kryo for the Structured Streaming runner (#39939)
e671af6c779 is described below
commit e671af6c77966058d40eea7d939f74656ee0d159
Author: Tobias Kaymak <[email protected]>
AuthorDate: Tue Sep 1 16:38:50 2026 +0200
[Spark][#36841] Register Spark's streaming internals with Kryo for the
Structured Streaming runner (#39939)
* [Spark 4] Register Spark's streaming internals with Kryo
A Structured Streaming query serializes two Spark internals behind the
runner's back, so streaming pipelines fail spark.kryo.registrationRequired
without these registrations:
- StateSchemaMetadata is broadcast for every transformWithState query,
hit on the very first micro-batch of any pipeline using Beam state or
timers.
- MemoryWriterCommitMessage is the memory sink's commit message, nested
inside the already registered DataWritingSparkTaskResult.
Both are registered by name because the shared base also compiles against
Spark 3, where neither class exists, and with a JavaSerializer so their
whole Scala object graph is covered without tracking Spark's internal
field layout across versions. Neither is on a hot path.
The registration call sits at the end of the registrator on purpose: Kryo
auto assigns ids sequentially, so appending these conditional, by-name
registrations keeps the auto assigned ids of everything above identical
on Spark 3 and Spark 4 classpaths.
SparkKryoRegistratorStreamingTest locks down both registrations and the
id parity.
---
.../SparkKryoRegistratorStreamingTest.java | 130 +++++++++++++++++++++
.../translation/SparkSessionFactory.java | 36 +++++-
2 files changed, 165 insertions(+), 1 deletion(-)
diff --git
a/runners/spark/4/src/test/java/org/apache/beam/runners/spark/structuredstreaming/translation/SparkKryoRegistratorStreamingTest.java
b/runners/spark/4/src/test/java/org/apache/beam/runners/spark/structuredstreaming/translation/SparkKryoRegistratorStreamingTest.java
new file mode 100644
index 00000000000..c37d726afb4
--- /dev/null
+++
b/runners/spark/4/src/test/java/org/apache/beam/runners/spark/structuredstreaming/translation/SparkKryoRegistratorStreamingTest.java
@@ -0,0 +1,130 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package org.apache.beam.runners.spark.structuredstreaming.translation;
+
+import static
org.apache.beam.runners.spark.structuredstreaming.translation.utils.ScalaInterop.seqOf;
+import static
org.apache.beam.runners.spark.structuredstreaming.translation.utils.ScalaInterop.tuple;
+import static org.junit.Assert.assertEquals;
+import static org.junit.Assert.assertNotNull;
+
+import com.esotericsoftware.kryo.Kryo;
+import com.esotericsoftware.kryo.io.Input;
+import com.esotericsoftware.kryo.io.Output;
+import java.io.ByteArrayInputStream;
+import java.io.ByteArrayOutputStream;
+import org.apache.beam.runners.spark.StreamingTest;
+import org.apache.spark.SparkConf;
+import org.apache.spark.serializer.KryoSerializer;
+import org.apache.spark.sql.Row;
+import org.apache.spark.sql.RowFactory;
+import
org.apache.spark.sql.execution.streaming.sources.MemoryWriterCommitMessage;
+import org.apache.spark.sql.execution.streaming.state.StateSchemaMetadata;
+import org.apache.spark.sql.execution.streaming.state.StateSchemaMetadataKey;
+import org.apache.spark.sql.execution.streaming.state.StateSchemaMetadataValue;
+import org.apache.spark.sql.types.DataTypes;
+import org.apache.spark.sql.types.StructType;
+import org.junit.Test;
+import org.junit.experimental.categories.Category;
+import org.junit.runner.RunWith;
+import org.junit.runners.JUnit4;
+import scala.collection.immutable.Map$;
+
+/**
+ * Guards the Spark 4 streaming entries of {@link
SparkSessionFactory.SparkKryoRegistrator}. The
+ * registrator references them by name, so a rename in a future Spark version
would silently drop
+ * the registration and only surface as a streaming query dying on its first
micro-batch. This test
+ * names the classes at compile time, turning that failure mode into a compile
error.
+ *
+ * @see SparkSessionFactory.SparkKryoRegistrator
+ */
+@RunWith(JUnit4.class)
+@Category(StreamingTest.class)
+public class SparkKryoRegistratorStreamingTest {
+
+ /** A Kryo configured exactly the way the runner configures it, strict
registration included. */
+ private static Kryo strictKryo() {
+ SparkConf conf =
+ new SparkConf(false)
+ .set("spark.serializer", KryoSerializer.class.getName())
+ .set("spark.kryo.registrationRequired", "true")
+ .set(
+ "spark.kryo.registrator",
SparkSessionFactory.SparkKryoRegistrator.class.getName());
+ return new KryoSerializer(conf).newKryo();
+ }
+
+ private static Object roundTrip(Kryo kryo, Object value) {
+ ByteArrayOutputStream bytes = new ByteArrayOutputStream();
+ try (Output output = new Output(bytes)) {
+ kryo.writeClassAndObject(output, value);
+ }
+ try (Input input = new Input(new
ByteArrayInputStream(bytes.toByteArray()))) {
+ return kryo.readClassAndObject(input);
+ }
+ }
+
+ /**
+ * Spark 4 broadcasts a {@link StateSchemaMetadata} to the executors for
every {@code
+ * transformWithState} query, so this is the registration that decides
whether a Beam streaming
+ * pipeline with state or timers runs at all under {@code
spark.kryo.registrationRequired=true}.
+ *
+ * <p>The instance below is deliberately not empty. It carries the nested
{@code StructType} and
+ * {@code org.apache.avro.Schema} that make the difference between a
registration that only
+ * survives a trivial payload and one that survives a real one.
+ */
+ @Test
+ public void stateSchemaMetadataRoundTripsWithRegistrationRequired() {
+ StructType sqlSchema =
+ new StructType().add("key", DataTypes.StringType).add("value",
DataTypes.BinaryType, false);
+ StateSchemaMetadataKey key = new StateSchemaMetadataKey("default", (short)
1, true);
+ StateSchemaMetadataValue value =
+ new StateSchemaMetadataValue(
+ sqlSchema,
org.apache.avro.Schema.create(org.apache.avro.Schema.Type.STRING));
+ StateSchemaMetadata metadata =
+ new StateSchemaMetadata(Map$.MODULE$.from(seqOf(tuple(key, value))));
+
+ Kryo kryo = strictKryo();
+ assertNotNull(
+ "StateSchemaMetadata must be registered, see SparkKryoRegistrator",
+ kryo.getRegistration(StateSchemaMetadata.class));
+
+ StateSchemaMetadata back = (StateSchemaMetadata) roundTrip(kryo, metadata);
+ assertEquals(1, back.activeSchemas().size());
+ assertEquals(value, back.activeSchemas().apply(key));
+ }
+
+ /**
+ * The commit message of Spark's {@code memory} sink, nested inside the
already registered {@code
+ * DataWritingSparkTaskResult}. The runner writes to {@code noop}, but the
{@code memory} sink is
+ * the obvious thing to reach for when inspecting a query, and it used to
fail on batch 0.
+ */
+ @Test
+ public void memoryWriterCommitMessageRoundTripsWithRegistrationRequired() {
+ Row row = RowFactory.create("a", 1);
+ MemoryWriterCommitMessage message = new MemoryWriterCommitMessage(3,
seqOf(row));
+
+ Kryo kryo = strictKryo();
+ assertNotNull(
+ "MemoryWriterCommitMessage must be registered, see
SparkKryoRegistrator",
+ kryo.getRegistration(MemoryWriterCommitMessage.class));
+
+ MemoryWriterCommitMessage back = (MemoryWriterCommitMessage)
roundTrip(kryo, message);
+ assertEquals(3, back.partition());
+ assertEquals(1, back.data().size());
+ assertEquals(row, back.data().apply(0));
+ }
+}
diff --git
a/runners/spark/src/main/java/org/apache/beam/runners/spark/structuredstreaming/translation/SparkSessionFactory.java
b/runners/spark/src/main/java/org/apache/beam/runners/spark/structuredstreaming/translation/SparkSessionFactory.java
index 822d1871b12..148188bb15a 100644
---
a/runners/spark/src/main/java/org/apache/beam/runners/spark/structuredstreaming/translation/SparkSessionFactory.java
+++
b/runners/spark/src/main/java/org/apache/beam/runners/spark/structuredstreaming/translation/SparkSessionFactory.java
@@ -22,6 +22,7 @@ import static
org.apache.commons.lang3.StringUtils.substringBetween;
import static org.apache.commons.lang3.math.NumberUtils.toInt;
import com.esotericsoftware.kryo.Kryo;
+import com.esotericsoftware.kryo.Serializer;
import com.esotericsoftware.kryo.serializers.JavaSerializer;
import java.util.ArrayList;
import java.util.Collection;
@@ -288,11 +289,44 @@ public class SparkSessionFactory {
kryo.register(CoGbkResultSchema.class);
kryo.register(TupleTag.class);
kryo.register(TupleTagList.class);
+
+ // Streaming internals that only exist as of Spark 4, registered by name
so this shared
+ // source also compiles against Spark 3. Must stay last so the auto
assigned ids of the
+ // registrations above are identical in the Spark 3 and Spark 4
artifacts.
+ registerSparkStreamingInternals(kryo);
+ }
+
+ /**
+ * Registers the internals a Structured Streaming query serializes behind
the runner's back, so
+ * streaming pipelines work with {@code
spark.kryo.registrationRequired=true}: {@code
+ * StateSchemaMetadata} (broadcast for every {@code transformWithState}
query) and {@code
+ * MemoryWriterCommitMessage} (the {@code memory} sink's commit message,
nested inside the
+ * already registered {@link DataWritingSparkTaskResult}). A {@link
JavaSerializer} covers their
+ * whole Scala object graph without tracking Spark's field layout; neither
is on a hot path.
+ */
+ private void registerSparkStreamingInternals(Kryo kryo) {
+ tryToRegister(
+ kryo,
+ "org.apache.spark.sql.execution.streaming.state.StateSchemaMetadata",
+ new JavaSerializer());
+ tryToRegister(
+ kryo,
+
"org.apache.spark.sql.execution.streaming.sources.MemoryWriterCommitMessage",
+ new JavaSerializer());
}
private void tryToRegister(Kryo kryo, String className) {
+ tryToRegister(kryo, className, null);
+ }
+
+ private void tryToRegister(Kryo kryo, String className, @Nullable
Serializer<?> serializer) {
try {
- kryo.register(Class.forName(className));
+ Class<?> cls = Class.forName(className);
+ if (serializer == null) {
+ kryo.register(cls);
+ } else {
+ kryo.register(cls, serializer);
+ }
} catch (ClassNotFoundException e) {
LOG.info("Class {}} was not found on classpath", className);
}