pvillard31 commented on code in PR #11499: URL: https://github.com/apache/nifi/pull/11499#discussion_r4093682489
########## nifi-extension-bundles/nifi-protobuf-bundle/nifi-protobuf-services/src/main/java/org/apache/nifi/services/protobuf/WriteProtobufResultWithExternalSchema.java: ########## @@ -0,0 +1,137 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.nifi.services.protobuf; + +import com.squareup.wire.schema.Schema; +import org.apache.nifi.schemaregistry.services.MessageIndexWriter; +import org.apache.nifi.schemaregistry.services.MessageName; +import org.apache.nifi.schemaregistry.services.SchemaDefinition; +import org.apache.nifi.schemaregistry.services.SchemaReferenceWriter; +import org.apache.nifi.serialization.AbstractRecordSetWriter; +import org.apache.nifi.serialization.record.Record; +import org.apache.nifi.serialization.record.RecordSchema; +import org.apache.nifi.services.protobuf.converter.ProtobufDataSerializer; + +import java.io.BufferedOutputStream; +import java.io.IOException; +import java.io.OutputStream; +import java.util.Map; + +/** + * Writes Records as Protocol Buffers binary content. When a {@link SchemaReferenceWriter} is + * configured, a Confluent wire-format header (magic byte and schema identifier) is written first; + * when a {@link MessageIndexWriter} is configured, the Confluent message index array follows the + * header. The serialized Protobuf payload is written last. + * <p> + * The Confluent framing (header and message index) is written once at the beginning of the record + * set, mirroring {@code WriteAvroResultWithExternalSchema}; the typical Confluent use case writes a + * single message per FlowFile. + */ +public class WriteProtobufResultWithExternalSchema extends AbstractRecordSetWriter { + + private final RecordSchema recordSchema; + private final SchemaDefinition schemaDefinition; + private final MessageName messageName; + private final SchemaReferenceWriter schemaReferenceWriter; + private final MessageIndexWriter messageIndexWriter; + private final Map<String, String> variables; + private final ProtobufDataSerializer serializer; + private final OutputStream buffered; + private boolean closed = false; + + public WriteProtobufResultWithExternalSchema(final Schema schema, + final MessageName messageName, + final RecordSchema recordSchema, + final SchemaDefinition schemaDefinition, + final SchemaReferenceWriter schemaReferenceWriter, + final MessageIndexWriter messageIndexWriter, + final Map<String, String> variables, + final OutputStream out) { + super(out); + this.recordSchema = recordSchema; + this.schemaDefinition = schemaDefinition; + this.messageName = messageName; + this.schemaReferenceWriter = schemaReferenceWriter; + this.messageIndexWriter = messageIndexWriter; + this.variables = variables; + this.buffered = new BufferedOutputStream(out); + this.serializer = new ProtobufDataSerializer(schema, messageName.getFullyQualifiedName()); + } + + @Override + protected void onBeginRecordSet() throws IOException { + writeConfluentFraming(buffered); + } + + @Override + protected Map<String, String> onFinishRecordSet() throws IOException { Review Comment: Should this return schemaReferenceWriter.getAttributes(recordSchema) when a Schema Reference Writer is configured, so attribute-based schema references are not discarded? ########## nifi-extension-bundles/nifi-protobuf-bundle/nifi-protobuf-services/src/main/java/org/apache/nifi/services/protobuf/converter/ProtobufDataSerializer.java: ########## @@ -0,0 +1,300 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.nifi.services.protobuf.converter; + +import com.google.protobuf.CodedOutputStream; +import com.squareup.wire.schema.EnumConstant; +import com.squareup.wire.schema.EnumType; +import com.squareup.wire.schema.Field; +import com.squareup.wire.schema.MessageType; +import com.squareup.wire.schema.OneOf; +import com.squareup.wire.schema.ProtoType; +import com.squareup.wire.schema.Schema; +import org.apache.nifi.serialization.record.Record; +import org.apache.nifi.serialization.record.util.DataTypeUtils; +import org.apache.nifi.services.protobuf.FieldType; + +import java.io.ByteArrayOutputStream; +import java.io.IOException; +import java.math.BigInteger; +import java.util.ArrayList; +import java.util.List; +import java.util.Map; +import java.util.Objects; + +/** + * Serializes a NiFi {@link Record} into Protocol Buffers binary payload using a Square Wire + * {@link Schema}. This is the write-side inverse of {@code ProtobufDataConverter}: it walks the + * declared fields of the target message type and encodes each present Record value to a + * {@link CodedOutputStream} following the Protocol Buffers wire format. + * <p> + * This class has no dependency on the NiFi framework and can be exercised directly with plain + * unit tests. + */ +public class ProtobufDataSerializer { + + private static final int WIRETYPE_VARINT = 0; + private static final int WIRETYPE_FIXED64 = 1; + private static final int WIRETYPE_LENGTH_DELIMITED = 2; + private static final int WIRETYPE_FIXED32 = 5; + + private static final int MAP_KEY_TAG = 1; + private static final int MAP_VALUE_TAG = 2; + + private final Schema schema; + private final String rootMessageType; + + public ProtobufDataSerializer(final Schema schema, final String rootMessageType) { + this.schema = schema; + this.rootMessageType = rootMessageType; + } + + /** + * Serializes the provided Record into Protocol Buffers binary format for the configured root + * message type. + * + * @param record the record to serialize + * @return the serialized protobuf payload + * @throws IOException if the record cannot be encoded + */ + public byte[] serialize(final Record record) throws IOException { + final MessageType messageType = (MessageType) schema.getType(rootMessageType); + Objects.requireNonNull(messageType, String.format("Message with name [%s] not found in the provided proto files", rootMessageType)); + + return serializeMessage(messageType, record); + } + + private byte[] serializeMessage(final MessageType messageType, final Record record) throws IOException { + final ByteArrayOutputStream output = new ByteArrayOutputStream(); + final CodedOutputStream codedOutput = CodedOutputStream.newInstance(output); + + for (final Field field : messageType.getDeclaredFields()) { + writeField(codedOutput, field, record.getValue(field.getName())); + } + for (final Field field : messageType.getExtensionFields()) { + writeField(codedOutput, field, record.getValue(field.getName())); + } + for (final OneOf oneOf : messageType.getOneOfs()) { + for (final Field field : oneOf.getFields()) { + writeField(codedOutput, field, record.getValue(field.getName())); + } + } + Review Comment: Should a missing proto2 required field fail serialization, or should the writer reject proto2 schemas if only proto3 output is supported? ########## nifi-extension-bundles/nifi-protobuf-bundle/nifi-protobuf-services/src/main/java/org/apache/nifi/services/protobuf/converter/ProtobufDataSerializer.java: ########## @@ -0,0 +1,300 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.nifi.services.protobuf.converter; + +import com.google.protobuf.CodedOutputStream; +import com.squareup.wire.schema.EnumConstant; +import com.squareup.wire.schema.EnumType; +import com.squareup.wire.schema.Field; +import com.squareup.wire.schema.MessageType; +import com.squareup.wire.schema.OneOf; +import com.squareup.wire.schema.ProtoType; +import com.squareup.wire.schema.Schema; +import org.apache.nifi.serialization.record.Record; +import org.apache.nifi.serialization.record.util.DataTypeUtils; +import org.apache.nifi.services.protobuf.FieldType; + +import java.io.ByteArrayOutputStream; +import java.io.IOException; +import java.math.BigInteger; +import java.util.ArrayList; +import java.util.List; +import java.util.Map; +import java.util.Objects; + +/** + * Serializes a NiFi {@link Record} into Protocol Buffers binary payload using a Square Wire + * {@link Schema}. This is the write-side inverse of {@code ProtobufDataConverter}: it walks the + * declared fields of the target message type and encodes each present Record value to a + * {@link CodedOutputStream} following the Protocol Buffers wire format. + * <p> + * This class has no dependency on the NiFi framework and can be exercised directly with plain + * unit tests. + */ +public class ProtobufDataSerializer { + + private static final int WIRETYPE_VARINT = 0; + private static final int WIRETYPE_FIXED64 = 1; + private static final int WIRETYPE_LENGTH_DELIMITED = 2; + private static final int WIRETYPE_FIXED32 = 5; + + private static final int MAP_KEY_TAG = 1; + private static final int MAP_VALUE_TAG = 2; + + private final Schema schema; + private final String rootMessageType; + + public ProtobufDataSerializer(final Schema schema, final String rootMessageType) { + this.schema = schema; + this.rootMessageType = rootMessageType; + } + + /** + * Serializes the provided Record into Protocol Buffers binary format for the configured root + * message type. + * + * @param record the record to serialize + * @return the serialized protobuf payload + * @throws IOException if the record cannot be encoded + */ + public byte[] serialize(final Record record) throws IOException { + final MessageType messageType = (MessageType) schema.getType(rootMessageType); + Objects.requireNonNull(messageType, String.format("Message with name [%s] not found in the provided proto files", rootMessageType)); + + return serializeMessage(messageType, record); + } + + private byte[] serializeMessage(final MessageType messageType, final Record record) throws IOException { + final ByteArrayOutputStream output = new ByteArrayOutputStream(); + final CodedOutputStream codedOutput = CodedOutputStream.newInstance(output); + + for (final Field field : messageType.getDeclaredFields()) { + writeField(codedOutput, field, record.getValue(field.getName())); + } + for (final Field field : messageType.getExtensionFields()) { + writeField(codedOutput, field, record.getValue(field.getName())); + } + for (final OneOf oneOf : messageType.getOneOfs()) { + for (final Field field : oneOf.getFields()) { + writeField(codedOutput, field, record.getValue(field.getName())); + } + } + + codedOutput.flush(); + return output.toByteArray(); + } + + private void writeField(final CodedOutputStream output, final Field field, final Object value) throws IOException { + if (value == null) { + return; + } + + final int tag = field.getTag(); + final ProtoType protoType = field.getType(); + + if (protoType.isMap()) { + writeMap(output, tag, protoType, value); + } else if (field.isRepeated()) { + writeRepeated(output, tag, protoType, value); + } else { + writeSingleValue(output, tag, protoType, value); + } + } + + private void writeRepeated(final CodedOutputStream output, final int tag, final ProtoType protoType, final Object value) throws IOException { + final Object[] values = toArray(value); + + if (isPackable(protoType)) { + // proto3 packs repeated scalar and enum fields into a single length-delimited entry + final ByteArrayOutputStream packedBytes = new ByteArrayOutputStream(); + final CodedOutputStream packedOutput = CodedOutputStream.newInstance(packedBytes); + for (final Object element : values) { + writeScalarOrEnumValueNoTag(packedOutput, protoType, element); + } + packedOutput.flush(); + + output.writeTag(tag, WIRETYPE_LENGTH_DELIMITED); + final byte[] packed = packedBytes.toByteArray(); + output.writeUInt32NoTag(packed.length); + output.writeRawBytes(packed); + } else { + // repeated messages, strings and bytes are written as separate length-delimited entries + for (final Object element : values) { + writeSingleValue(output, tag, protoType, element); + } + } + } + + private void writeSingleValue(final CodedOutputStream output, final int tag, final ProtoType protoType, final Object value) throws IOException { + if (protoType.isScalar()) { + final FieldType fieldType = FieldType.findValue(protoType.getSimpleName()); + output.writeTag(tag, wireTypeFor(fieldType)); + writeScalarValueNoTag(output, fieldType, value); + return; + } + + if (schema.getType(protoType) instanceof EnumType) { + output.writeTag(tag, WIRETYPE_VARINT); + output.writeEnumNoTag(enumTag(protoType, value)); + return; + } + + // nested message + final MessageType messageType = (MessageType) schema.getType(protoType); + Objects.requireNonNull(messageType, String.format("Message type with name [%s] not found in the provided proto files", protoType)); + final byte[] nested = serializeMessage(messageType, toRecord(value, protoType)); + output.writeTag(tag, WIRETYPE_LENGTH_DELIMITED); + output.writeUInt32NoTag(nested.length); + output.writeRawBytes(nested); + } + + private void writeMap(final CodedOutputStream output, final int tag, final ProtoType protoType, final Object value) throws IOException { + if (!(value instanceof final Map<?, ?> map)) { + throw new IOException(String.format("Expected a Map value for map field but received [%s]", value.getClass())); + } + + final ProtoType keyType = protoType.getKeyType(); + final ProtoType valueType = protoType.getValueType(); + + for (final Map.Entry<?, ?> entry : map.entrySet()) { + final ByteArrayOutputStream entryBytes = new ByteArrayOutputStream(); + final CodedOutputStream entryOutput = CodedOutputStream.newInstance(entryBytes); + + writeSingleValue(entryOutput, MAP_KEY_TAG, keyType, entry.getKey()); Review Comment: Should null map keys be rejected with a clear IOException before scalar conversion? ########## nifi-extension-bundles/nifi-protobuf-bundle/nifi-protobuf-services/src/main/java/org/apache/nifi/services/protobuf/StandardProtobufWriter.java: ########## @@ -0,0 +1,356 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.nifi.services.protobuf; + +import com.squareup.wire.schema.Schema; +import org.apache.nifi.annotation.documentation.CapabilityDescription; +import org.apache.nifi.annotation.documentation.Tags; +import org.apache.nifi.annotation.lifecycle.OnEnabled; +import org.apache.nifi.components.AllowableValue; +import org.apache.nifi.components.DescribedValue; +import org.apache.nifi.components.PropertyDescriptor; +import org.apache.nifi.components.PropertyValue; +import org.apache.nifi.context.PropertyContext; +import org.apache.nifi.controller.AbstractControllerService; +import org.apache.nifi.controller.ConfigurationContext; +import org.apache.nifi.controller.ControllerServiceInitializationContext; +import org.apache.nifi.logging.ComponentLog; +import org.apache.nifi.processor.util.StandardValidators; +import org.apache.nifi.reporting.InitializationException; +import org.apache.nifi.schema.access.SchemaNotFoundException; +import org.apache.nifi.schemaregistry.services.MessageIndexWriter; +import org.apache.nifi.schemaregistry.services.MessageName; +import org.apache.nifi.schemaregistry.services.MessageNameResolver; +import org.apache.nifi.schemaregistry.services.SchemaDefinition; +import org.apache.nifi.schemaregistry.services.SchemaReferenceWriter; +import org.apache.nifi.schemaregistry.services.SchemaRegistry; +import org.apache.nifi.schemaregistry.services.StandardMessageNameFactory; +import org.apache.nifi.schemaregistry.services.StandardSchemaDefinition; +import org.apache.nifi.serialization.RecordSetWriter; +import org.apache.nifi.serialization.RecordSetWriterFactory; +import org.apache.nifi.serialization.SchemaRegistryService; +import org.apache.nifi.serialization.SimpleRecordSchema; +import org.apache.nifi.serialization.record.RecordSchema; +import org.apache.nifi.serialization.record.SchemaIdentifier; +import org.apache.nifi.services.protobuf.schema.ProtoSchemaParser; + +import java.io.ByteArrayInputStream; +import java.io.IOException; +import java.io.InputStream; +import java.io.OutputStream; +import java.nio.charset.StandardCharsets; +import java.security.MessageDigest; +import java.security.NoSuchAlgorithmException; +import java.util.ArrayList; +import java.util.HexFormat; +import java.util.List; +import java.util.Map; + +import static org.apache.nifi.expression.ExpressionLanguageScope.FLOWFILE_ATTRIBUTES; +import static org.apache.nifi.schema.access.SchemaAccessUtils.SCHEMA_ACCESS_STRATEGY; +import static org.apache.nifi.schema.access.SchemaAccessUtils.SCHEMA_BRANCH_NAME; +import static org.apache.nifi.schema.access.SchemaAccessUtils.SCHEMA_NAME; Review Comment: Should @SeeAlso(StandardProtobufReader.class) be added so users can discover the companion reader? ########## nifi-extension-bundles/nifi-confluent-platform-bundle/nifi-confluent-protobuf-message-index-writer/src/main/java/org/apache/nifi/confluent/schemaregistry/ProtobufMessageIndexEncoder.java: ########## @@ -0,0 +1,113 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.nifi.confluent.schemaregistry; + +import org.apache.nifi.confluent.schema.ProtobufMessageSchema; +import org.apache.nifi.confluent.schema.VarintUtils; +import org.apache.nifi.schemaregistry.services.MessageName; + +import java.io.ByteArrayOutputStream; +import java.util.ArrayList; +import java.util.List; + +import static java.lang.String.format; + +/** + * Computes the Confluent wire format message index path for a target message within a parsed + * Protobuf schema, and encodes it to bytes. This is the inverse of the message index decoding + * performed by {@code ConfluentProtobufMessageNameResolver}: given a fully qualified message + * name, it locates the path of declaration-order indexes leading to that message and encodes it + * as zigzag varints, applying the single-byte {@code 0x00} optimization for the common case of + * the first root message. + * <p> + * <a href="https://docs.confluent.io/platform/current/schema-registry/fundamentals/serdes-develop/index.html#wire-format">See the Confluent protobuf wire format.</a> + * <p> + * This class has no dependency on the NiFi framework and can be exercised directly with plain + * unit tests. + */ +final class ProtobufMessageIndexEncoder { + + private static final byte[] FIRST_ROOT_MESSAGE_INDEX = {0x00}; + + private ProtobufMessageIndexEncoder() { + } + + /** + * Encodes the message index path for the given message name within the given schema. + * + * @param rootMessages the root messages of the parsed Protobuf schema, in declaration order + * @param messageName the target message name to locate + * @return the encoded message index bytes + * @throws IllegalStateException if the message name cannot be located within the schema + */ + static byte[] encode(final List<ProtobufMessageSchema> rootMessages, final MessageName messageName) { + final List<Integer> messageIndexPath = findMessageIndexPath(rootMessages, messageName); + + if (messageIndexPath.size() == 1 && messageIndexPath.getFirst() == 0) { + return FIRST_ROOT_MESSAGE_INDEX; Review Comment: Should this return a copy of FIRST_ROOT_MESSAGE_INDEX so callers cannot mutate shared cached data? ########## nifi-extension-bundles/nifi-protobuf-bundle/nifi-protobuf-services/src/main/java/org/apache/nifi/services/protobuf/StandardProtobufWriter.java: ########## @@ -0,0 +1,356 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package org.apache.nifi.services.protobuf; + +import com.squareup.wire.schema.Schema; +import org.apache.nifi.annotation.documentation.CapabilityDescription; +import org.apache.nifi.annotation.documentation.Tags; +import org.apache.nifi.annotation.lifecycle.OnEnabled; +import org.apache.nifi.components.AllowableValue; +import org.apache.nifi.components.DescribedValue; +import org.apache.nifi.components.PropertyDescriptor; +import org.apache.nifi.components.PropertyValue; +import org.apache.nifi.context.PropertyContext; +import org.apache.nifi.controller.AbstractControllerService; +import org.apache.nifi.controller.ConfigurationContext; +import org.apache.nifi.controller.ControllerServiceInitializationContext; +import org.apache.nifi.logging.ComponentLog; +import org.apache.nifi.processor.util.StandardValidators; +import org.apache.nifi.reporting.InitializationException; +import org.apache.nifi.schema.access.SchemaNotFoundException; +import org.apache.nifi.schemaregistry.services.MessageIndexWriter; +import org.apache.nifi.schemaregistry.services.MessageName; +import org.apache.nifi.schemaregistry.services.MessageNameResolver; +import org.apache.nifi.schemaregistry.services.SchemaDefinition; +import org.apache.nifi.schemaregistry.services.SchemaReferenceWriter; +import org.apache.nifi.schemaregistry.services.SchemaRegistry; +import org.apache.nifi.schemaregistry.services.StandardMessageNameFactory; +import org.apache.nifi.schemaregistry.services.StandardSchemaDefinition; +import org.apache.nifi.serialization.RecordSetWriter; +import org.apache.nifi.serialization.RecordSetWriterFactory; +import org.apache.nifi.serialization.SchemaRegistryService; +import org.apache.nifi.serialization.SimpleRecordSchema; +import org.apache.nifi.serialization.record.RecordSchema; +import org.apache.nifi.serialization.record.SchemaIdentifier; +import org.apache.nifi.services.protobuf.schema.ProtoSchemaParser; + +import java.io.ByteArrayInputStream; +import java.io.IOException; +import java.io.InputStream; +import java.io.OutputStream; +import java.nio.charset.StandardCharsets; +import java.security.MessageDigest; +import java.security.NoSuchAlgorithmException; +import java.util.ArrayList; +import java.util.HexFormat; +import java.util.List; +import java.util.Map; + +import static org.apache.nifi.expression.ExpressionLanguageScope.FLOWFILE_ATTRIBUTES; +import static org.apache.nifi.schema.access.SchemaAccessUtils.SCHEMA_ACCESS_STRATEGY; +import static org.apache.nifi.schema.access.SchemaAccessUtils.SCHEMA_BRANCH_NAME; +import static org.apache.nifi.schema.access.SchemaAccessUtils.SCHEMA_NAME; +import static org.apache.nifi.schema.access.SchemaAccessUtils.SCHEMA_NAME_PROPERTY; +import static org.apache.nifi.schema.access.SchemaAccessUtils.SCHEMA_REFERENCE_READER; +import static org.apache.nifi.schema.access.SchemaAccessUtils.SCHEMA_REGISTRY; +import static org.apache.nifi.schema.access.SchemaAccessUtils.SCHEMA_TEXT; +import static org.apache.nifi.schema.access.SchemaAccessUtils.SCHEMA_TEXT_PROPERTY; +import static org.apache.nifi.schema.access.SchemaAccessUtils.SCHEMA_VERSION; +import static org.apache.nifi.services.protobuf.StandardProtobufWriter.MessageNameResolverStrategy.MESSAGE_NAME_PROPERTY; + +@Tags({"protobuf", "record", "writer", "serializer", "confluent"}) +@CapabilityDescription(""" + Serializes NiFi Records into Protocol Buffers binary format. \ + Supports inline schema text and schema registry lookup for determining the Proto schema. \ + When a Schema Reference Writer is configured, a Confluent wire-format header is written; when a \ + Message Index Writer is also configured, the Confluent message index array is written after the header. \ + The target Proto message name can be determined statically using the 'Message Name' property, \ + or dynamically using a Message Name Resolver service. + A single record is written per FlowFile, since concatenated Protocol Buffers messages cannot be delimited. \ + The 'google.protobuf.Any' well-known type is not expanded on write; a Record derived from an Any-typed message \ + is serialized as an ordinary nested message rather than being re-wrapped as an Any.""") +public class StandardProtobufWriter extends SchemaRegistryService implements RecordSetWriterFactory { + + public static final PropertyDescriptor MESSAGE_NAME_RESOLUTION_STRATEGY = new PropertyDescriptor.Builder() + .name("Message Name Resolution Strategy") + .description("Strategy for determining the Protocol Buffers message name for serialization") + .required(true) + .allowableValues(MESSAGE_NAME_PROPERTY, MessageNameResolverStrategy.MESSAGE_NAME_RESOLVER) + .defaultValue(MESSAGE_NAME_PROPERTY) + .build(); + + public static final PropertyDescriptor MESSAGE_NAME = new PropertyDescriptor.Builder() + .name("Message Name") + .description("Fully qualified name of the Protocol Buffers message including its package (eg. mypackage.MyMessage).") + .required(true) + .expressionLanguageSupported(FLOWFILE_ATTRIBUTES) + .dependsOn(MESSAGE_NAME_RESOLUTION_STRATEGY, MESSAGE_NAME_PROPERTY) + .addValidator(StandardValidators.NON_EMPTY_VALIDATOR) + .build(); + + public static final PropertyDescriptor MESSAGE_NAME_RESOLVER = new PropertyDescriptor.Builder() + .name("Message Name Resolver") + .description("Service that dynamically resolves Protocol Buffer message names from FlowFile attributes. " + + "On the write side the resolver is invoked with an empty content stream, so only resolvers that derive the " + + "message name from attributes are supported; resolvers that read the message name from message content " + + "(such as the Confluent wire-format resolver used on the read side) are not applicable here.") + .required(true) + .identifiesControllerService(MessageNameResolver.class) + .dependsOn(MESSAGE_NAME_RESOLUTION_STRATEGY, MessageNameResolverStrategy.MESSAGE_NAME_RESOLVER) + .build(); + + public static final PropertyDescriptor SCHEMA_REFERENCE_WRITER = new PropertyDescriptor.Builder() + .name("Schema Reference Writer") + .description("Service used to write schema reference information, such as a Confluent wire-format header, before the serialized Protobuf content. " + + "When not configured, plain Protobuf content is written without any header.") + .required(false) + .identifiesControllerService(SchemaReferenceWriter.class) + .build(); + + public static final PropertyDescriptor MESSAGE_INDEX_WRITER = new PropertyDescriptor.Builder() + .name("Message Index Writer") + .description("Service used to write the Confluent message index array identifying the target message within the schema, written after the Schema Reference Writer header. " + + "Applicable only when producing Confluent wire-format content.") + .required(false) + .identifiesControllerService(MessageIndexWriter.class) + .build(); + + private static final PropertyDescriptor PROTOBUF_SCHEMA_TEXT = new PropertyDescriptor.Builder() + .fromPropertyDescriptor(SCHEMA_TEXT) + .required(true) + .clearValidators() + .addValidator(StandardValidators.NON_EMPTY_VALIDATOR) + .defaultValue("${proto.schema}") + .description("The text of a Proto 3 formatted Schema") + .build(); + + private static final String PROTO_EXTENSION = ".proto"; + + private static final InputStream EMPTY_INPUT_STREAM = new ByteArrayInputStream(new byte[0]); + + private volatile ProtobufSchemaCompiler schemaCompiler; + private volatile MessageNameResolver messageNameResolver; + private volatile SchemaReferenceWriter schemaReferenceWriter; + private volatile MessageIndexWriter messageIndexWriter; + private volatile SchemaRegistry schemaRegistry; + private volatile String schemaAccessStrategyValue; + private volatile PropertyValue schemaText; + private volatile PropertyValue schemaName; + private volatile PropertyValue schemaBranchName; + private volatile PropertyValue schemaVersion; + + @OnEnabled + public void onEnabled(final ConfigurationContext context) { + super.storeSchemaAccessStrategy(context); + setupMessageNameResolver(context); + schemaAccessStrategyValue = context.getProperty(SCHEMA_ACCESS_STRATEGY).getValue(); + schemaRegistry = context.getProperty(SCHEMA_REGISTRY).asControllerService(SchemaRegistry.class); + schemaReferenceWriter = context.getProperty(SCHEMA_REFERENCE_WRITER).asControllerService(SchemaReferenceWriter.class); + messageIndexWriter = context.getProperty(MESSAGE_INDEX_WRITER).asControllerService(MessageIndexWriter.class); + schemaName = context.getProperty(SCHEMA_NAME); + schemaText = context.getProperty(SCHEMA_TEXT); + schemaBranchName = context.getProperty(SCHEMA_BRANCH_NAME); + schemaVersion = context.getProperty(SCHEMA_VERSION); + } + + @Override + protected void init(final ControllerServiceInitializationContext config) throws InitializationException { + super.init(config); + schemaCompiler = new ProtobufSchemaCompiler(getIdentifier(), getLogger()); + } + + @Override + public RecordSchema getSchema(final Map<String, String> variables, final RecordSchema readSchema) throws SchemaNotFoundException, IOException { + return createWriteContext(variables).recordSchema(); + } + + @Override + public RecordSetWriter createWriter(final ComponentLog logger, final RecordSchema schema, final OutputStream out, final Map<String, String> variables) throws SchemaNotFoundException, IOException { Review Comment: Should createWriter() use the supplied schema instead of resolving the registry schema again, so a moving latest version cannot change between getSchema() and writer creation? -- This is an automated message from the Apache Git Service. To respond to the message, please log on to GitHub and use the URL above to go to the specific comment. To unsubscribe, e-mail: [email protected] For queries about this service, please contact Infrastructure at: [email protected]
